diff --git a/test/auto-decide-current-declaration.test.ts b/test/auto-decide-current-declaration.test.ts deleted file mode 100644 index ab6753d83..000000000 --- a/test/auto-decide-current-declaration.test.ts +++ /dev/null @@ -1,198 +0,0 @@ -import { expect, test } from 'bun:test'; -import { findNativeAutoDecision } from './helpers/native-auto-decide'; -import capture from './fixtures/auto-decide-current-declaration-6aef.json'; - -const clone = () => structuredClone(capture.retry) as any; -const decide = (f: any) => findNativeAutoDecision(f.transcript, f.tools, f.options); -const message = (f: any) => f.transcript.assistantMessages.find((m: any) => - m.timestamp === '2026-09-16T23:23:28.931Z'); -const use = (f: any) => f.tools.find((e: any) => e.kind === 'use' && - e.input?.command?.includes('gstack-skill-start')); -const ack = (f: any) => f.tools.find((e: any) => e.kind === 'result' && e.toolUseId === use(f).toolUseId); - -test('actual current Decision declaration completes the retained owned retry', () => { - const f = clone(); - const result = decide(f); - expect(result?.option).toBe('HOLD SCOPE'); - expect(result?.annotation).toBe(message(f).text); - expect(result?.stateRecord).toEqual(f.options.stateEvidence.records[0]); - expect(result?.preambleToolUseId).toBe(use(f).toolUseId); - expect(result?.skillToolUseId).toBeUndefined(); - expect(result?.questionLogToolUseId).toBeUndefined(); -}); - -test('literal first declaration form is supported by the authenticated retry context', () => { - const f = clone(); - // This checks representation only. The first attempt's state was deleted; - // transplanting its text grants no first-attempt ownership or verdict credit. - message(f).text = capture.firstDeclaration.text; - expect(decide(f)?.option).toBe('HOLD SCOPE'); - delete f.options.stateEvidence; - f.tools = []; - expect(decide(f)).toBeNull(); -}); - -const modes = ['HOLD SCOPE', 'SCOPE EXPANSION', 'SELECTIVE EXPANSION', 'SCOPE REDUCTION']; -for (const mode of modes) { - for (const label of ['Decision', 'Decision: review mode is', 'Mode']) { - for (const target of ['for this draft.', 'for the current review (saved preference).']) { - const text = `${label}${label.includes(':') ? '' : ':'} ${mode} ${target}`; - test(`one current full mode owns its target clause: ${text}`, () => { - const f = clone(); - Object.assign(f.options.stateEvidence.records[0], { user_choice: mode, recommended: mode }); - message(f).text = text; - expect(decide(f)?.option).toBe(mode); - }); - } - } -} - -const invalidDeclarations = [ - 'Decision: HOLD SCOPELESS for this draft.', - 'Decision: HOLD for this draft.', - 'Decision: SCOPE for this draft.', - 'Decision: SELECTIVE for this draft.', - 'Decision: review mode is CUSTOM MODE for this draft.', - 'Decision: HOLD SCOPE?', - 'Decision: HOLD SCOPE for', - 'Decision: HOLD SCOPE for (', - 'Decision: HOLD SCOPE for this draft (unfinished.', - 'Decision: HOLD SCOPE for this draft (unbalanced)).', - 'Decision: HOLD SCOPE for this draft or SCOPE EXPANSION.', - 'Decision: HOLD SCOPE for this draft. Instead choose SCOPE REDUCTION.', - 'Decision: HOLD SCOPE for this draft; SELECTIVE_EXPANSION.', - 'Decision: HOLD SCOPE for this draft, if approved.', - 'Decision: HOLD SCOPE for this draft, pending approval.', - 'Decision: HOLD SCOPE for this draft, not yet selected.', - 'Decision: HOLD SCOPE for this draft, withdrawn.', - 'Decision: HOLD SCOPE for this draft; the selected mode is not HOLD SCOPE.', - 'Decision: HOLD SCOPE for plan-eng-review.', - 'Decision: HOLD SCOPE for another draft.', - 'Decision: HOLD SCOPE for a future review.', - 'Decision: HOLD SCOPE for this future review.', - 'Decision pending: HOLD SCOPE for this draft.', - 'Decision: review mode is not selected.', - 'Decision: not HOLD SCOPE for this draft.', -]; -for (const text of invalidDeclarations) { - test(`unsupported current declaration cannot complete a decision: ${text}`, () => { - const f = clone(); message(f).text = text; - expect(decide(f)).toBeNull(); - }); - test(`unsupported current declaration retracts the earlier decision: ${text}`, () => { - const f = clone(); message(f).text += `\n\nCorrection: ${text}`; - expect(decide(f)).toBeNull(); - }); -} - -for (const prefix of ['> ', ' ', '"', '`']) { - test(`quoted current-mode syntax does not declare or retract: ${JSON.stringify(prefix)}`, () => { - const f = clone(); - const quote = (value: string) => prefix + value + (['"', '`'].includes(prefix) ? prefix : ''); - message(f).text = quote('Decision: HOLD SCOPE for this draft.'); - expect(decide(f)).toBeNull(); - message(f).text = clone().transcript.assistantMessages.at(-1).text + '\n\n' + - quote('Decision: SCOPE EXPANSION for this draft.'); - expect(decide(f)?.option).toBe('HOLD SCOPE'); - }); -} -for (const text of [ - 'Example:\nDecision: HOLD SCOPE for this draft.', - 'Historical transcript:\nDecision: HOLD SCOPE for this draft.', - 'Previous decision:\nDecision: HOLD SCOPE for this draft.', - '```text\nDecision: HOLD SCOPE for this draft.\n```', - 'If approved, Decision: HOLD SCOPE for this draft.', - 'Not a decision: HOLD SCOPE for this draft.', -]) test(`unasserted declaration provides no mode: ${JSON.stringify(text)}`, () => { - const f = clone(); message(f).text = text; - expect(decide(f)).toBeNull(); -}); - -for (const label of ['Decision', 'Decision: review mode is', 'Mode']) { - test(`later matching current declaration retains the owned mode: ${label}`, () => { - const f = clone(); message(f).text += `\n\nUpdate: ${label}${label.includes(':') ? '' : ':'} HOLD SCOPE for this draft.`; - expect(decide(f)?.option).toBe('HOLD SCOPE'); - }); - test(`later conflicting current declaration retracts the owned mode: ${label}`, () => { - const f = clone(); message(f).text += `\n\nUpdate: ${label}${label.includes(':') ? '' : ':'} SCOPE EXPANSION for this draft.`; - expect(decide(f)).toBeNull(); - }); -} - -test('a separately scoped non-mode decision does not retract the review mode', () => { - const f = clone(); message(f).text += '\n\nDecision: publish the audit log.'; - expect(decide(f)?.option).toBe('HOLD SCOPE'); -}); - -test('the completed native preference and log ACK retain their independent authority', () => { - const f = clone(); delete f.options.stateEvidence; - const result = decide(f); - expect(result?.option).toBe('HOLD SCOPE'); - expect(result?.stateRecord).toBeUndefined(); - expect(result?.preferenceToolUseId).toBeDefined(); - expect(result?.questionLogToolUseId).toBeDefined(); -}); - -test('target-clause capitalization and a negative non-mode explanation remain valid', () => { - const f = clone(); message(f).text = 'Decision: HOLD SCOPE For this draft, not for implementation.'; - expect(decide(f)?.option).toBe('HOLD SCOPE'); -}); - -test('a named target must match the complete audit target, including dotted identifiers', () => { - const f = clone(); - f.options.stateEvidence.records[0].question_summary = 'Select review mode for PLAN.md'; - message(f).text = 'Decision: HOLD SCOPE for PLAN.md.'; - expect(decide(f)?.option).toBe('HOLD SCOPE'); - message(f).text = 'Decision: HOLD SCOPE for PLAN.other.'; - expect(decide(f)).toBeNull(); -}); - -for (const target of ['a future review', 'the previous review', 'another draft', 'the next invocation', 'an example plan']) { - test(`even a matching audit cannot make an explicitly noncurrent target current: ${target}`, () => { - const f = clone(); - f.options.stateEvidence.records[0].question_summary = `Select review mode for ${target}`; - message(f).text = `Decision: HOLD SCOPE for ${target}.`; - expect(decide(f)).toBeNull(); - }); -} -for (const target of ['future.md', 'previous-review.md', 'another.plan.md']) { - test(`owned literal filename remains a current target: ${target}`, () => { - const f = clone(); - f.options.stateEvidence.records[0].question_summary = `Select review mode for ${target}`; - message(f).text = `Decision: HOLD SCOPE for ${target}.`; - expect(decide(f)?.option).toBe('HOLD SCOPE'); - }); -} - -for (const [name, mutate] of Object.entries({ - 'missing state and native log ACK': (f: any) => { - delete f.options.stateEvidence; - const log = f.tools.find((e: any) => e.kind === 'use' && e.input?.command?.includes('gstack-question-log')); - f.tools = f.tools.filter((e: any) => e.kind !== 'result' || e.toolUseId !== log.toolUseId); - }, - 'empty owned log': (f: any) => { f.options.stateEvidence.records = []; }, - 'duplicate owned log': (f: any) => { f.options.stateEvidence.records.push({ ...f.options.stateEvidence.records[0] }); }, - 'wrong preference and native check': (f: any) => { - f.options.stateEvidence.preference = 'always-ask'; - const check = f.tools.find((e: any) => e.kind === 'use' && e.input?.command?.includes('gstack-question-preference')); - f.tools.find((e: any) => e.kind === 'result' && e.toolUseId === check.toolUseId).content = 'ASK\nEXIT: 0'; - }, - 'foreign log session': (f: any) => { f.options.stateEvidence.records[0].session_id = 'foreign'; }, - 'foreign log skill': (f: any) => { f.options.stateEvidence.records[0].skill = 'plan-eng-review'; }, - 'different logged choice': (f: any) => { f.options.stateEvidence.records[0].user_choice = 'SCOPE EXPANSION'; }, - 'different recommendation': (f: any) => { f.options.stateEvidence.records[0].recommended = 'SCOPE EXPANSION'; }, - 'nonautomatic log': (f: any) => { f.options.stateEvidence.records[0].auto_decided = false; }, - 'foreign native session': (f: any) => { f.options.sessionId = 'foreign'; }, - 'failed preamble': (f: any) => { ack(f).isError = true; }, - 'missing preamble ACK': (f: any) => { f.tools = f.tools.filter((e: any) => e !== ack(f)); }, - 'duplicate preamble ACK': (f: any) => { f.tools.push({ ...ack(f) }); }, - 'disabled tuning': (f: any) => { ack(f).content = ack(f).content.replace('QUESTION_TUNING: true', 'QUESTION_TUNING: false'); }, - 'old log': (f: any) => { f.options.stateEvidence.records[0].ts = new Date(f.options.commandStartedAt - 1).toISOString(); }, - 'future log': (f: any) => { f.options.stateEvidence.records[0].ts = new Date(f.options.now + 1).toISOString(); }, - 'public decision before log': (f: any) => { message(f).timestamp = new Date(Date.parse(f.options.stateEvidence.records[0].ts) - 1).toISOString(); }, - 'native question': (f: any) => { f.transcript.calls.push({ sessionId: f.options.sessionId }); }, - 'prose question': (f: any) => { f.options.proseQuestionObserved = true; }, - 'current withdrawal': (f: any) => { message(f).text += '\n\nI withdraw this decision.'; }, -})) test(`current Decision syntax retains ${name} rejection`, () => { - const f = clone(); mutate(f); expect(decide(f)).toBeNull(); -}); diff --git a/test/auto-decide-explanatory-mode.test.ts b/test/auto-decide-explanatory-mode.test.ts deleted file mode 100644 index ef15e6887..000000000 --- a/test/auto-decide-explanatory-mode.test.ts +++ /dev/null @@ -1,263 +0,0 @@ -import { expect, test } from 'bun:test'; -import { findNativeAutoDecision } from './helpers/native-auto-decide'; -import capture from './fixtures/auto-decide-explanatory-mode-043a.json'; -import captured749 from './fixtures/auto-decide-explanatory-mode-749df.json'; -import annotations from './fixtures/native-auto-decide-ag.json'; - -const clone = () => structuredClone(capture) as any; -const declaration = (f: any) => f.transcript.assistantMessages.find((m: any) => - m.text.startsWith('**Review mode: HOLD SCOPE** —')); -const decide = (f: any) => findNativeAutoDecision(f.transcript, f.tools, f.options); - -function witnessed() { - const f = clone(); - const use = f.tools.find((e: any) => e.input?.command?.includes('gstack-question-log')); - // Synthetic owned-file witness from the exact literal request. The original - // file was not retained; its failed paid attempt remains failed. - const record = JSON.parse(/gstack-question-log '(\{[^\n]*\})'/.exec(use.input.command)![1]!); - record.source = 'agent'; - record.ts = f.tools.find((e: any) => e.kind === 'result' && e.toolUseId === use.toolUseId).timestamp; - f.options.stateEvidence = { questionId: 'plan-ceo-review-mode', preference: 'never-ask', records: [record] }; - return f; -} - -test('original public events alone cannot authenticate the unretained owned append', () => { - expect(decide(clone())).toBeNull(); -}); - -test('exact completed announcement agrees with an authenticated owned append', () => { - const f = witnessed(); - const result = decide(f); - expect(result?.option).toBe('HOLD SCOPE'); - expect(result?.annotation).toBe(declaration(f).text); - expect(result?.stateRecord).toEqual(f.options.stateEvidence.records[0]); - expect(result?.questionLogToolUseId).toBeUndefined(); -}); - -const separators = ['. ', ', ', '; ', ': ', ' — ', ' – ', ' - ']; -for (const separator of separators) { - test(`complete mode with separated explanation ${JSON.stringify(separator)}`, () => { - const f = witnessed(); - declaration(f).text = `Review mode: HOLD SCOPE${separator}selected from the saved preference.`; - expect(decide(f)?.option).toBe('HOLD SCOPE'); - }); - test(`same later completed mode retains its explanation ${JSON.stringify(separator)}`, () => { - const f = witnessed(); - declaration(f).text += `\n\nMode decision completed: HOLD SCOPE${separator}selected from the saved preference.`; - expect(decide(f)?.option).toBe('HOLD SCOPE'); - }); - test(`different later completed mode withdraws the choice ${JSON.stringify(separator)}`, () => { - const f = witnessed(); - declaration(f).text += `\n\nCorrection: Mode: SCOPE EXPANSION${separator}selected from the saved preference.`; - expect(decide(f)).toBeNull(); - }); - test(`conditional explanation never completes the mode ${JSON.stringify(separator)}`, () => { - const f = witnessed(); - declaration(f).text = `Review mode: HOLD SCOPE${separator}if approved.`; - expect(decide(f)).toBeNull(); - }); - test(`later conditional explanation withdraws the choice ${JSON.stringify(separator)}`, () => { - const f = witnessed(); - declaration(f).text += `\n\nMode: HOLD SCOPE${separator}pending approval.`; - expect(decide(f)).toBeNull(); - }); -} - -for (const value of ['HOLD SCOPELESS', 'HOLD SCOPE SCOPE EXPANSION', 'HOLD SCOPE / SCOPE EXPANSION', - 'HOLD SCOPE?', 'HOLD SCOPE selected from my preference', 'HOLD SCOPE—if approved', 'HOLD SCOPE - ']) { - test(`incomplete or ambiguous mode is not a declaration: ${value}`, () => { - const f = witnessed(); declaration(f).text = `Mode: ${value}`; - expect(decide(f)).toBeNull(); - }); - test(`incomplete current field retracts a previous mode: ${value}`, () => { - const f = witnessed(); declaration(f).text += `\n\nMode: ${value}`; - expect(decide(f)).toBeNull(); - }); -} - -for (const [name, mutate] of Object.entries({ - 'missing append': (f: any) => { f.options.stateEvidence.records = []; }, - 'foreign session': (f: any) => { f.options.stateEvidence.records[0].session_id = 'foreign'; }, - 'different logged mode': (f: any) => { f.options.stateEvidence.records[0].user_choice = 'SCOPE EXPANSION'; }, - 'native question': (f: any) => { f.transcript.calls.push({ sessionId: f.options.sessionId }); }, - 'unfinished declaration': (f: any) => { declaration(f).text = 'Mode decision pending: HOLD SCOPE — saved preference.'; }, - 'withdrawn decision': (f: any) => { declaration(f).text += '\n\nI withdraw this decision.'; }, - 'quoted declaration': (f: any) => { declaration(f).text = '> Mode: HOLD SCOPE — saved preference.'; }, - 'example declaration': (f: any) => { declaration(f).text = 'Example:\nMode: HOLD SCOPE — saved preference.'; }, -})) test(`explanatory announcement still rejects ${name}`, () => { - const f = witnessed(); mutate(f); expect(decide(f)).toBeNull(); -}); - -test('quoted historical correction does not withdraw the current completed mode', () => { - const f = witnessed(); - declaration(f).text += '\n\n> Mode: SCOPE EXPANSION — a historical example.'; - expect(decide(f)?.option).toBe('HOLD SCOPE'); -}); - -test('generic Skill annotations retain their existing non-CEO mode vocabulary', () => { - const f = structuredClone(annotations.attempts[0]) as any; - f.options.skillName = 'office-hours'; - f.tools.find((e: any) => e.kind === 'use' && e.name === 'Skill').input.skill = 'office-hours'; - const message = f.transcript.assistantMessages.find((m: any) => m.text.startsWith('Auto-decided')); - message.text = 'Auto-decided workflow → **Builder** (your preference). Change with /plan-tune.\n\nMode: Builder (saved preference).'; - expect(decide(f)?.option).toBe('Builder'); - message.text += '\n\nMode: Startup (saved preference).'; - expect(decide(f)).toBeNull(); -}); - -test('retained retry messages alone cannot authenticate missing tool and file evidence', () => { - const retry = capture.retryObservation; - expect(findNativeAutoDecision(retry.transcript, [], retry.options)).toBeNull(); -}); - -test('exact retry prose accepts the optional decision label in an owned context', () => { - const f = witnessed(); - // Only the text is replayed. Session/time and owned witness belong to the - // first fixture; this is not a reconstruction or promotion of the retry. - declaration(f).text = capture.retryObservation.transcript.assistantMessages.find(m => - m.text.startsWith('**Mode decision:'))!.text; - expect(decide(f)?.option).toBe('HOLD SCOPE'); -}); - -for (const field of ['Mode', 'Mode decision', 'Review mode', 'Review mode decision']) { - test(`a completed field does not require a separate status word: ${field}`, () => { - const f = witnessed(); declaration(f).text = `${field}: HOLD SCOPE (saved preference).`; - expect(decide(f)?.option).toBe('HOLD SCOPE'); - }); - test(`a later matching field does not withdraw its choice: ${field}`, () => { - const f = witnessed(); declaration(f).text += `\n\n${field}: HOLD SCOPE (saved preference).`; - expect(decide(f)?.option).toBe('HOLD SCOPE'); - }); - test(`a conflicting later field still withdraws its choice: ${field}`, () => { - const f = witnessed(); declaration(f).text += `\n\n${field}: SCOPE EXPANSION (saved preference).`; - expect(decide(f)).toBeNull(); - }); -} - -const clone749 = () => structuredClone(captured749) as any; -const declaration749 = (f: any) => f.transcript.assistantMessages.find((m: any) => - m.timestamp === '2026-09-16T12:13:02.513Z'); - -test('actual 749 public declaration agrees with its retained owned append', () => { - // Exact public tools, final declaration and owned log were captured while the - // paid observer was still waiting. This free replay does not promote that run. - const f = clone749(); - const result = decide(f); - expect(result?.option).toBe('HOLD SCOPE'); - expect(result?.annotation).toBe(declaration749(f).text); - expect(result?.stateRecord).toEqual(f.options.stateEvidence.records[0]); -}); - -const explanatoryTails = [ - ' (saved preference; no prompt required). The review remains paused.', - ' (saved preference (confirmed for this project); no prompt required). The review remains paused.', - ' (saved preference), recorded for this invocation.', - ' (saved preference): recorded for this invocation.', - ' (saved preference) — recorded for this invocation.', - ' (saved preference)\nThe review remains paused.', - '. Selected from the saved preference (recorded).', - '; selected from the saved preference (recorded).', -]; -for (const tail of explanatoryTails) { - test(`balanced explanation with following prose is a complete declaration: ${JSON.stringify(tail)}`, () => { - const f = clone749(); declaration749(f).text = `Mode decision: HOLD SCOPE${tail}`; - expect(decide(f)?.option).toBe('HOLD SCOPE'); - }); - test(`matching later explanation preserves the current decision: ${JSON.stringify(tail)}`, () => { - const f = clone749(); declaration749(f).text += `\n\nMode: HOLD SCOPE${tail}`; - expect(decide(f)?.option).toBe('HOLD SCOPE'); - }); - test(`conflicting later explanation withdraws the current decision: ${JSON.stringify(tail)}`, () => { - const f = clone749(); declaration749(f).text += `\n\nMode: SCOPE EXPANSION${tail}`; - expect(decide(f)).toBeNull(); - }); -} - -const incompleteFields = [ - 'Mode: HOLD SCOPE (saved preference; recorded.', - 'Mode: HOLD SCOPE (saved preference (recorded).', - 'Mode: HOLD SCOPE (saved preference)). Recorded.', - 'Mode: HOLD SCOPE (saved preference)SCOPE EXPANSION', - 'Mode: HOLD SCOPE. A following explanation (unfinished.', - 'Mode: HOLD SCOPE (saved preference). If approved.', - 'Mode: HOLD SCOPE (saved preference (if approved)). Recorded.', - 'Mode: HOLD SCOPE (saved preference). Not yet selected.', - 'Mode: HOLD SCOPE (saved preference). I did not auto-decide the review mode.', - 'Mode: HOLD SCOPE (saved preference). This decision is withdrawn.', - 'Mode decision pending: HOLD SCOPE (saved preference). Recorded.', - 'Mode decision tentative: HOLD SCOPE (saved preference). Recorded.', - 'Mode: CUSTOM MODE (saved preference). Recorded.', - 'Mode: HOLD SCOPE / SCOPE EXPANSION (saved preference). Recorded.', - 'Mode: HOLD SCOPE (saved preference).\nMode decision pending: HOLD SCOPE', -]; -for (const field of incompleteFields) { - test(`explanatory prose cannot complete an unsupported field: ${JSON.stringify(field)}`, () => { - const f = clone749(); declaration749(f).text = field; - expect(decide(f)).toBeNull(); - }); - test(`later unsupported field retracts the earlier decision: ${JSON.stringify(field)}`, () => { - const f = clone749(); declaration749(f).text += `\n\n${field}`; - expect(decide(f)).toBeNull(); - }); -} - -for (const [name, mutate] of Object.entries({ - 'missing owned append': (f: any) => { f.options.stateEvidence.records = []; }, - 'foreign owned session': (f: any) => { f.options.stateEvidence.records[0].session_id = 'foreign'; }, - 'duplicate owned append': (f: any) => { f.options.stateEvidence.records.push({ ...f.options.stateEvidence.records[0] }); }, - 'conflicting logged choice': (f: any) => { f.options.stateEvidence.records[0].user_choice = 'SCOPE EXPANSION'; }, - 'failed preamble': (f: any) => { - const preamble = f.tools.find((e: any) => e.kind === 'use' && e.input?.command?.includes('gstack-skill-start')); - f.tools.find((e: any) => e.kind === 'result' && e.toolUseId === preamble.toolUseId).isError = true; - }, - 'native question': (f: any) => { f.transcript.calls.push({ sessionId: f.options.sessionId }); }, - 'declaration before owned append': (f: any) => { declaration749(f).timestamp = '2026-09-16T12:12:00.000Z'; }, - 'quoted declaration': (f: any) => { declaration749(f).text = '> Mode: HOLD SCOPE (saved preference). Recorded.'; }, - 'example declaration': (f: any) => { declaration749(f).text = 'Example:\nMode: HOLD SCOPE (saved preference). Recorded.'; }, -})) test(`captured explanatory mode still requires ${name}`, () => { - const f = clone749(); mutate(f); expect(decide(f)).toBeNull(); -}); - -const reviewModes = ['HOLD SCOPE', 'SCOPE EXPANSION', 'SELECTIVE EXPANSION', 'SCOPE REDUCTION']; -for (const mode of reviewModes) { - for (const tail of [' (saved preference). Recorded for this invocation.', - ' (saved preference (confirmed); recorded). No further mode decision.', - ` (saved preference). ${mode} is recorded for this invocation.`]) { - test(`each complete review mode supports an unambiguous explanatory suffix: ${mode}${tail}`, () => { - const f = clone749(); - Object.assign(f.options.stateEvidence.records[0], { user_choice: mode, recommended: mode }); - declaration749(f).text = `Mode: ${mode}${tail}`; - expect(decide(f)?.option).toBe(mode); - }); - } - for (const other of reviewModes.filter(value => value !== mode)) { - for (const connector of [' or ', ' versus ', ' vs. ', ' / ', ' | ', '; or ', ', choose ', - '. Alternatively, select ', ' — instead choose ', ' (otherwise choose ', ' rather than ']) { - test(`a second distinct mode in the suffix stays ambiguous: ${mode}${connector}${other}`, () => { - const f = clone749(); - Object.assign(f.options.stateEvidence.records[0], { user_choice: mode, recommended: mode }); - const tail = connector.startsWith(' (') ? ')' : ''; - declaration749(f).text = `Mode: ${mode} (saved preference)${connector}${other}${tail}`; - expect(decide(f)).toBeNull(); - }); - } - } -} - -test('alternate current mode spellings remain ambiguous after an explanatory parenthetical', () => { - for (const alternative of ['scope expansion', 'SCOPE_EXPANSION', 'SCOPE EXPANSION']) { - const f = clone749(); declaration749(f).text = `Mode: HOLD SCOPE (saved preference); ${alternative}`; - expect(decide(f)).toBeNull(); - } -}); - -test('generic annotation vocabulary retains its original parenthetical boundaries', () => { - for (const suffix of ['', ' or Startup', '; or Startup', ' versus Startup', '. Recorded for this invocation.']) { - const f = structuredClone(annotations.attempts[0]) as any; - f.options.skillName = 'office-hours'; - f.tools.find((e: any) => e.kind === 'use' && e.name === 'Skill').input.skill = 'office-hours'; - const message = f.transcript.assistantMessages.find((m: any) => m.text.startsWith('Auto-decided')); - message.text = `Auto-decided workflow → **Builder** (your preference). Change with /plan-tune.\n\nMode: Builder (saved preference)${suffix}`; - expect(decide(f)?.option ?? null).toBe(suffix ? null : 'Builder'); - } -}); diff --git a/test/auto-decide-recommendation-scope.test.ts b/test/auto-decide-recommendation-scope.test.ts deleted file mode 100644 index 0db938fc5..000000000 --- a/test/auto-decide-recommendation-scope.test.ts +++ /dev/null @@ -1,100 +0,0 @@ -import { expect, test } from 'bun:test'; -import { findNativeAutoDecision } from './helpers/native-auto-decide'; -import capture from './fixtures/auto-decide-recommendation-361c.json'; - -const clone = () => structuredClone(capture) as any; -const decide = (f: any) => findNativeAutoDecision(f.transcript, f.tools, f.options); -const message = (f: any) => f.transcript.assistantMessages.find((m: any) => - m.timestamp === '2026-09-17T02:23:11.495Z'); - -test('actual completed mode and recommendation commentary match the retained owned audit', () => { - const f = clone(), result = decide(f); - expect(result?.option).toBe('HOLD SCOPE'); - expect(result?.annotation).toBe(message(f).text); - expect(result?.stateRecord).toEqual(f.options.stateEvidence.records[0]); - expect(result?.preambleToolUseId).toBe('toolu_01Ni4b9NZeiexmcAz1jRTUa4'); -}); - -const modes = ['HOLD SCOPE', 'SCOPE EXPANSION', 'SELECTIVE EXPANSION', 'SCOPE REDUCTION']; -for (const mode of modes) for (const commentary of [ - 'recommendation would have been the same', - 'my recommendation might differ without the saved preference', - `the recommendation would still be ${mode}`, - 'our recommendation will remain unchanged', - 'recommendation stays the same unless the product context changes', -]) test(`completed ${mode} is separate from ${commentary}`, () => { - const f = clone(); - Object.assign(f.options.stateEvidence.records[0], { user_choice: mode, recommended: mode }); - message(f).text = `Decision: review mode is ${mode} (${commentary}).`; - expect(decide(f)?.option).toBe(mode); -}); - -for (const separator of ['; ', ', ', '. ', ' — ', ' – ', ' - ', ' (']) { - test(`recommendation assertion has an explicit boundary: ${JSON.stringify(separator)}`, () => { - const f = clone(); - message(f).text = `Mode: HOLD SCOPE${separator}recommendation would have been unchanged${separator === ' (' ? ')' : ''}.`; - expect(decide(f)?.option).toBe('HOLD SCOPE'); - }); -} - -const uncertain = [ - 'Mode: would choose HOLD SCOPE.', - 'Mode: HOLD SCOPE if approved.', - 'Mode: HOLD SCOPE unless you object.', - 'Mode: HOLD SCOPE (I will make this selection).', - 'Mode: HOLD SCOPE (this choice might change).', - 'Mode: HOLD SCOPE (recommendation would be the same; if approved).', - 'Mode: HOLD SCOPE (recommendation would be the same, unless you object).', - 'Mode: HOLD SCOPE (recommendation would be the same and I will select it later).', - 'Mode: HOLD SCOPE (recommendation would be the same but the choice might change).', - 'Mode: HOLD SCOPE (recommendation would be the same while we would still need approval).', - 'Mode: HOLD SCOPE (recommendation would be the same; selection is pending).', - 'Mode: HOLD SCOPE (recommendation says the decision would be conditional).', - 'Mode: HOLD SCOPE (recommendation would still be SCOPE EXPANSION).', - 'Mode: HOLD SCOPE (recommendation would be unchanged). Not yet selected.', - 'Mode: HOLD SCOPE (recommendation would be unchanged). This decision is withdrawn.', - 'Mode pending: HOLD SCOPE (recommendation would be unchanged).', - 'Mode: not HOLD SCOPE (recommendation would be unchanged).', - 'Mode: HOLD SCOPE for a future review (recommendation would be unchanged).', - 'Mode: HOLD SCOPE for another draft (recommendation would be unchanged).', - 'Mode: HOLD SCOPELESS (recommendation would be unchanged).', - 'Mode: HOLD SCOPE (recommendation would be unchanged.', -]; -for (const text of uncertain) { - test(`commentary does not authenticate an uncertain choice: ${text}`, () => { - const f = clone(); message(f).text = text; - expect(decide(f)).toBeNull(); - }); - test(`later uncertain choice retracts the original completed decision: ${text}`, () => { - const f = clone(); message(f).text += `\n\nCorrection: ${text}`; - expect(decide(f)).toBeNull(); - }); -} - -for (const [name, wrap] of [ - ['quoted', (s: string) => `> ${s}`], - ['indented', (s: string) => ` ${s}`], - ['fenced', (s: string) => `\`\`\`text\n${s}\n\`\`\``], - ['historical', (s: string) => `Previous decision:\n${s}`], - ['example', (s: string) => `Example:\n${s}`], -] as const) test(`recommendation commentary cannot authenticate ${name} declarations`, () => { - const f = clone(); message(f).text = wrap('Mode: HOLD SCOPE (recommendation would be unchanged).'); - expect(decide(f)).toBeNull(); -}); - -for (const [name, mutate] of Object.entries({ - 'missing owned log': (f: any) => { f.options.stateEvidence.records = []; }, - 'duplicate owned log': (f: any) => { f.options.stateEvidence.records.push({ ...f.options.stateEvidence.records[0] }); }, - 'foreign audit session': (f: any) => { f.options.stateEvidence.records[0].session_id = 'foreign'; }, - 'different selected choice': (f: any) => { f.options.stateEvidence.records[0].user_choice = 'SCOPE EXPANSION'; }, - 'different recommendation': (f: any) => { f.options.stateEvidence.records[0].recommended = 'SCOPE EXPANSION'; }, - 'nonautomatic record': (f: any) => { f.options.stateEvidence.records[0].auto_decided = false; }, - 'foreign native session': (f: any) => { f.options.sessionId = 'foreign'; }, - 'missing successful preamble': (f: any) => { f.tools = f.tools.filter((e: any) => e.toolUseId !== 'toolu_01Ni4b9NZeiexmcAz1jRTUa4'); }, - 'future audit': (f: any) => { f.options.stateEvidence.records[0].ts = new Date(f.options.now + 1).toISOString(); }, - 'decision before completed log': (f: any) => { message(f).timestamp = new Date(Date.parse(f.options.stateEvidence.records[0].ts) - 1).toISOString(); }, - 'surfaced native question': (f: any) => { f.transcript.calls.push({ sessionId: f.options.sessionId }); }, - 'surfaced prose question': (f: any) => { f.options.proseQuestionObserved = true; }, -})) test(`actual recommendation commentary retains ${name} rejection`, () => { - const f = clone(); mutate(f); expect(decide(f)).toBeNull(); -}); diff --git a/test/auto-decide-saved-ai.test.ts b/test/auto-decide-saved-ai.test.ts deleted file mode 100644 index 673b1f257..000000000 --- a/test/auto-decide-saved-ai.test.ts +++ /dev/null @@ -1,80 +0,0 @@ -import { expect, test } from 'bun:test'; -import { findNativeAutoDecision } from './helpers/native-auto-decide'; -import captured from './fixtures/auto-decide-saved-ai.json'; -const clone=()=>structuredClone(captured) as any; -const decision=(f=clone())=>findNativeAutoDecision(f.transcript,f.tools,f.options); -const message=(f:any)=>f.transcript.assistantMessages.find((m:any)=>m.text.includes('Auto-decided')); - -test('actual saved mode preference annotation is a completed native auto-decision',()=>{ - const f=clone(), result=decision(f); - expect(result).not.toBeNull(); - expect(result!.option).toBe('HOLD SCOPE'); - expect(message(f).text).toContain(result!.annotation); - expect(f.transcript.calls).toEqual([]); -}); - -test('saved preference is bound to this invoked skill and an agreeing current mode',()=>{ - for(const change of [ - (s:string)=>s.replace('`plan-ceo-review-mode`','`plan-design-review-mode`'), - (s:string)=>s.replace('`plan-ceo-review-mode`','`plan-ceo-review-routing`'), - (s:string)=>s.replace('**Review mode: HOLD SCOPE.**','**Review mode: SCOPE EXPANSION.**'), - (s:string)=>s.replace('**Review mode: HOLD SCOPE.**\n\n',''), - (s:string)=>s.replace('"Select review mode"','"Select report folder"'), - (s:string)=>s.replace('(your saved preference on','(a proposed preference on'), - (s:string)=>s.replace('Change with /plan-tune.',''), - (s:string)=>s.replace('Auto-decided','I will auto-decide'), - ]) {const f=clone();message(f).text=change(message(f).text);expect(decision(f)).toBeNull();} -}); - -test('prefixed examples, quotations and hypothetical notices do not assert a current choice',()=>{ - for(const change of [ - (s:string)=>'Example:\n\n'+s, - (s:string)=>'```text\n'+s+'\n```', - (s:string)=>s.split('\n').map(l=>'> '+l).join('\n'), - (s:string)=>s.replace('Heads-up from the preamble: unshipped work on this branch','Heads-up from the preamble: a hypothetical example'), - (s:string)=>s.replace('Auto-decided "Select',' Auto-decided "Select'), - (s:string)=>s.replace('Auto-decided "Select','If approved, Auto-decided "Select'), - ]) {const f=clone();message(f).text=change(message(f).text);expect(decision(f)).toBeNull();} -}); - -test('failed loads, foreign sessions, actual questions and later withdrawals retain precedence',()=>{ - for(const mutate of [ - (f:any)=>{f.options.sessionId='foreign';}, - (f:any)=>{const use=f.tools.find((t:any)=>t.kind==='use'&&t.name==='Skill');f.tools.find((t:any)=>t.kind==='result'&&t.toolUseId===use.toolUseId).isError=true;}, - (f:any)=>{f.transcript.calls.push({sessionId:f.options.sessionId,toolUseId:'actual-question'});}, - (f:any)=>{f.options.now=Date.parse(message(f).timestamp)-1;}, - (f:any)=>{message(f).text+='\n\nCorrection: I withdraw this decision.';}, - (f:any)=>{message(f).text+='\n\n**Review mode: SCOPE EXPANSION.**';}, - ]) {const f=clone();mutate(f);expect(decision(f)).toBeNull();} -}); - -import { E2E_TOUCHFILES } from './helpers/touchfiles-data'; -test('new evidence inputs retain every native observation caller',()=>{ - const owners=['plan-ceo-review-plan-mode','plan-eng-review-plan-mode','plan-design-review-plan-mode','plan-devex-review-plan-mode','plan-mode-no-op','auto-decide-preserved']; - for(const file of ['test/auto-decide-saved-ai.test.ts','test/fixtures/auto-decide-saved-ai.json','test/fixtures/auto-decide-retry-ai.json']) - expect(Object.entries(E2E_TOUCHFILES).filter(([,paths])=>paths.includes(file)).map(([name])=>name)).toEqual(owners); -}); - -import retry from './fixtures/auto-decide-retry-ai.json'; -test('actual retry mode-decision heading retains its own annotation, excluding prior foreign text',()=>{ - const f:any=structuredClone(retry), actual=decision(f); - expect(actual).not.toBeNull();expect(actual!.sessionId).toBe(f.options.sessionId); - expect(actual!.option).toBe('HOLD SCOPE'); - expect(actual!.annotation).toContain('(your preference)'); - const own=f.transcript.assistantMessages.filter((m:any)=>m.sessionId===f.options.sessionId); - f.transcript.assistantMessages=f.transcript.assistantMessages.filter((m:any)=>m.sessionId!==f.options.sessionId); - expect(decision(f)).toBeNull();expect(own.length).toBeGreaterThan(0); -}); - -test('retry heading cannot supply a hypothetical, different decision, or withdrawn selection',()=>{ - for(const change of [ - (s:string)=>'Example:\n\n'+s, - (s:string)=>s.replace('Review mode for the deterministic','Review mode for the hypothetical'), - (s:string)=>s.replace('D1 — Review mode','D1 — Report destination'), - (s:string)=>s.replace('Auto-decided "Review mode:', 'Auto-decided "Report destination:'), - (s:string)=>s.replace('→ **HOLD SCOPE**','→ **Save a file**'), - (s:string)=>s+'\n\nCorrection: I withdraw this selection.', - (s:string)=>s+'\n\n**Review mode: SCOPE EXPANSION.**', - (s:string)=>s.replace('Heads-up from gstack: there is unshipped work on this branch','Heads-up from gstack: here is an example'), - ]) {const f:any=structuredClone(retry),m=f.transcript.assistantMessages.find((m:any)=>m.sessionId===f.options.sessionId&&m.text.includes('Auto-decided'));m.text=change(m.text);expect(decision(f)).toBeNull();} -}); diff --git a/test/auto-decide-structured.test.ts b/test/auto-decide-structured.test.ts deleted file mode 100644 index 94ba75ead..000000000 --- a/test/auto-decide-structured.test.ts +++ /dev/null @@ -1,221 +0,0 @@ -import { expect, test } from 'bun:test'; -import { findNativeAutoDecision } from './helpers/native-auto-decide'; -import capture from './fixtures/auto-decide-structured-77.json'; -const clone = () => structuredClone(capture) as any; -const decision = (f = clone()) => findNativeAutoDecision(f.transcript, f.tools, f.options); -test('actual slash expansion with completed preference log and current mode is an auto-decision', () => { - const f = clone(); - expect(f.tools.some((e: any) => e.name === 'Skill')).toBe(false); - expect(f.transcript.calls).toEqual([]); - const result = decision(f); - expect(result).not.toBeNull(); - expect(result!.option).toBe('HOLD SCOPE'); -}); - -const use = (f: any, name: string) => f.tools.find((e: any) => e.kind === 'use' && e.input?.command?.includes(name)); -const ack = (f: any, request: any) => f.tools.find((e: any) => e.kind === 'result' && e.toolUseId === request.toolUseId); -const modeMessage = (f: any) => f.transcript.assistantMessages.find((m: any) => m.text.includes('**Mode:')); -const changeLog = (f: any, modify: (log: any) => void) => { - const request = use(f, 'gstack-question-log'), match = /'(\{.*\})'/.exec(request.input.command)!; - const value = JSON.parse(match[1]!); modify(value); - request.input.command = request.input.command.replace(match[1], JSON.stringify(value)); -}; - -for (const [label, mutate] of Object.entries({ - 'missing transcript': (f: any) => { f.transcript.status = 'missing'; }, - 'foreign owned session': (f: any) => { f.options.sessionId = 'foreign'; }, - 'wrong invoked skill': (f: any) => { f.options.skillName = 'plan-eng-review'; }, - 'pre-command evidence': (f: any) => { f.options.commandStartedAt = Date.parse(modeMessage(f).timestamp); }, - 'future final statement': (f: any) => { f.options.now = Date.parse(modeMessage(f).timestamp) - 1; }, - 'invalid final timestamp': (f: any) => { modeMessage(f).timestamp = 'invalid'; }, - 'native question': (f: any) => { f.transcript.calls.push({ sessionId: f.options.sessionId, toolUseId: 'asked' }); }, - 'malformed native question tool': (f: any) => { f.tools.push({ ...use(f, 'gstack-question-log'), toolUseId: 'asked', name: 'mcp__ask__AskUserQuestion', input: {} }); }, - 'earlier visible prose question': (f: any) => { f.options.proseQuestionObserved = true; }, - 'public reply request': (f: any) => { modeMessage(f).text += '\nReply with A or B.'; }, - 'public option list': (f: any) => { modeMessage(f).text += '\nA) Hold scope\nB) Expand scope'; }, - 'no preamble': (f: any) => { const request = use(f, 'gstack-skill-start'); f.tools = f.tools.filter((e: any) => e.toolUseId !== request.toolUseId); }, - 'preamble failed': (f: any) => { ack(f, use(f, 'gstack-skill-start')).isError = true; }, - 'preamble missing ACK': (f: any) => { const request = use(f, 'gstack-skill-start'); f.tools = f.tools.filter((e: any) => e !== ack(f, request)); }, - 'preamble duplicate': (f: any) => { f.tools.push({ ...use(f, 'gstack-skill-start') }); }, - 'wrong preamble skill': (f: any) => { use(f, 'gstack-skill-start').input.command = use(f, 'gstack-skill-start').input.command.replace('--skill "plan-ceo-review"', '--skill "plan-eng-review"'); }, - 'question tuning disabled': (f: any) => { const result = ack(f, use(f, 'gstack-skill-start')); result.content = result.content.replace('QUESTION_TUNING: true', 'QUESTION_TUNING: false'); }, - 'ambiguous preamble session': (f: any) => { ack(f, use(f, 'gstack-skill-start')).content = 'SKILL_START_PROTO: 1\nQUESTION_TUNING: true\nSESSION_ID: duplicate\n' + ack(f, use(f, 'gstack-skill-start')).content; }, - 'nonzero preference': (f: any) => { ack(f, use(f, 'gstack-question-preference')).content = 'AUTO_DECIDE\nEXIT: 1'; }, - 'ASK preference': (f: any) => { ack(f, use(f, 'gstack-question-preference')).content = 'ASK\nEXIT: 0'; }, - 'preference error': (f: any) => { ack(f, use(f, 'gstack-question-preference')).isError = true; }, - 'wrong preference id': (f: any) => { use(f, 'gstack-question-preference').input.command = use(f, 'gstack-question-preference').input.command.replace('--check "plan-ceo-review-mode"', '--check "plan-ceo-review-other"'); }, - 'no preference check': (f: any) => { const request = use(f, 'gstack-question-preference'); f.tools = f.tools.filter((e: any) => e.toolUseId !== request.toolUseId); }, - 'unacknowledged log': (f: any) => { const request = use(f, 'gstack-question-log'); f.tools = f.tools.filter((e: any) => e !== ack(f, request)); }, - 'failed log': (f: any) => { ack(f, use(f, 'gstack-question-log')).isError = true; }, - 'fallback log result': (f: any) => { ack(f, use(f, 'gstack-question-log')).content = 'log unavailable (best-effort)'; }, - 'wrong log session': (f: any) => changeLog(f, log => { log.session_id = 'foreign'; }), - 'wrong log skill': (f: any) => changeLog(f, log => { log.skill = 'plan-eng-review'; }), - 'wrong log question id': (f: any) => changeLog(f, log => { log.question_id = 'plan-ceo-review-scope'; }), - 'nonautomatic log': (f: any) => changeLog(f, log => { log.auto_decided = false; }), - 'string automatic flag': (f: any) => changeLog(f, log => { log.auto_decided = 'true'; }), - 'unmatched recommendation': (f: any) => changeLog(f, log => { log.recommended = 'SCOPE_EXPANSION'; }), - 'different logged mode': (f: any) => changeLog(f, log => { log.recommended = log.user_choice = 'SCOPE_EXPANSION'; }), - 'arbitrary logged value': (f: any) => changeLog(f, log => { log.recommended = log.user_choice = 'APPROVE_SCOPE'; }), - 'nondecision summary': (f: any) => changeLog(f, log => { log.question_summary = ''; }), - 'later checked preference': (f: any) => { ack(f, use(f, 'gstack-question-preference')).timestamp = modeMessage(f).timestamp; }, - 'mode before log ACK': (f: any) => { modeMessage(f).timestamp = use(f, 'gstack-question-log').timestamp; }, - 'reversed log ACK': (f: any) => { ack(f, use(f, 'gstack-question-log')).timestamp = use(f, 'gstack-question-preference').timestamp; }, - 'duplicate log ACK': (f: any) => { f.tools.push({ ...ack(f, use(f, 'gstack-question-log')) }); }, - 'foreign log ACK': (f: any) => { ack(f, use(f, 'gstack-question-log')).sessionId = 'foreign'; }, - 'missing current statement': (f: any) => { modeMessage(f).text = 'Done. Waiting for your next instruction.'; }, -})) test(`structured current mode rejects ${label}`, () => { - const f = clone(); mutate(f); expect(decision(f)).toBeNull(); -}); - -for (const name of ['gstack-skill-start', 'gstack-question-preference', 'gstack-question-log']) { - for (const [label, change] of Object.entries({ - 'echoed source': (s: string) => `echo '${s.replaceAll("'", "'\\''")}'`, - 'conditional command': (s: string) => `false && ${s}`, - 'commented source': (s: string) => `# ${s}`, - 'extra prefix command': (s: string) => `true; ${s}`, - 'extra suffix command': (s: string) => `${s}; true`, - 'command substitution': (s: string) => `echo "$(${s})"`, - })) test(`${name} cannot authenticate ${label}`, () => { - const f = clone(); use(f, name).input.command = change(use(f, name).input.command); expect(decision(f)).toBeNull(); - }); -} - -for (const [label, text] of Object.entries({ - 'plain current field': 'Mode: HOLD SCOPE.', - 'parenthetical explanation with punctuation': 'Mode: HOLD SCOPE (saved preference, confirmed).', - 'parenthetical review explanation': '**Review mode: HOLD SCOPE (saved preference; confirmed).**', - 'current review field': '**Review mode: HOLD SCOPE.**', - 'compact completion': '**STATUS: DONE**\n\nMode: HOLD SCOPE', - 'bullet conclusion': 'The requested routing decision is complete.\n\n- **Mode: HOLD SCOPE**, using the saved preference.\n\nThe substantive review is deferred.', - 'quoted historical contradiction': 'Mode: HOLD SCOPE.\n\nEarlier example: "Review mode: SCOPE EXPANSION."', -})) test(`completed structured log supports ${label} without exact annotation prose`, () => { - const f = clone(); modeMessage(f).text = text; - const result = decision(f); expect(result?.option).toBe('HOLD SCOPE'); - expect(result?.skillToolUseId).toBeUndefined(); - expect(result?.preambleToolUseId).toBe(use(f, 'gstack-skill-start').toolUseId); - expect(result?.annotation).toBe(text); -}); - -for (const text of [ - '> Mode: HOLD SCOPE.', ' Mode: HOLD SCOPE.', '`Mode: HOLD SCOPE.`', - '```text\nMode: HOLD SCOPE.\n```', 'Example:\n\nMode: HOLD SCOPE.', - 'Previous transcript:\n\nMode: HOLD SCOPE.', 'If approved, Mode: HOLD SCOPE.', - 'Mode: HOLD SCOPE, if you approve.', 'Mode: HOLD SCOPE, pending approval.', - 'Mode: HOLD SCOPE?', 'Mode: HOLD SCOPELESS.', - 'Mode: HOLD SCOPE (withdrawn).', 'Mode: HOLD SCOPE (retracted).', - 'Mode: HOLD SCOPE.\n\nMode: HOLD SCOPE (pending approval).', - 'Mode: HOLD SCOPE.\n\nCorrection: I withdraw this decision.', - 'Mode: HOLD SCOPE.\n\nI did not auto-decide the review mode.', - 'Mode: HOLD SCOPE.\n\nCorrection: Mode: SCOPE EXPANSION.', - 'Mode: HOLD SCOPE.\n\nMode: SCOPE EXPANSION.', -]) test(`quoted, conditional or withdrawn mode has no completed choice: ${JSON.stringify(text)}`, () => { - const f = clone(); modeMessage(f).text = text; expect(decision(f)).toBeNull(); -}); - -test('the same command contracts also support direct literal invocations and quiet ACKs', () => { - const f = clone(); - use(f, 'gstack-skill-start').input.command = '"$HOME/.claude/skills/gstack/bin/gstack-skill-start" --model claude --skill plan-ceo-review --parent-pid "$PPID"'; - use(f, 'gstack-question-preference').input.command = '~/.claude/skills/gstack/bin/gstack-question-preference --check plan-ceo-review-mode'; - ack(f, use(f, 'gstack-question-preference')).content = 'AUTO_DECIDE\n'; - use(f, 'gstack-question-log').input.command = use(f, 'gstack-question-log').input.command.split(' 2>/dev/null')[0]; - ack(f, use(f, 'gstack-question-log')).content = ''; - expect(decision(f)?.option).toBe('HOLD SCOPE'); -}); - -import priorAnnotation from './fixtures/auto-decide-saved-ai.json'; -for (const status of ['undecided', 'not selected', 'pending approval', 'none']) { - test(`later Review mode: ${status} withdraws both existing annotation and structured decision`, () => { - const previous: any = structuredClone(priorAnnotation); - previous.transcript.assistantMessages.find((m: any) => m.text.includes('Auto-decided')).text += `\n\nReview mode: ${status}.`; - expect(findNativeAutoDecision(previous.transcript, previous.tools, previous.options)).toBeNull(); - const f = clone(); modeMessage(f).text += `\n\nReview mode: ${status}.`; - expect(decision(f)).toBeNull(); - }); - test(`later Mode: ${status} withdraws a structured decision`, () => { - const f = clone(); modeMessage(f).text += `\n\n- **Mode: ${status}.**`; - expect(decision(f)).toBeNull(); - }); -} - -for (const name of ['gstack-question-preference', 'gstack-question-log']) test(`${name} cannot borrow an earlier success after a contradictory current call`, () => { - const f = clone(), request = structuredClone(use(f, name)), result = structuredClone(ack(f, request)); - request.toolUseId += '-later'; result.toolUseId = request.toolUseId; - request.timestamp = result.timestamp = new Date(Date.parse(modeMessage(f).timestamp) - 1).toISOString(); - if (name === 'gstack-question-preference') result.content = 'ASK\nEXIT: 0'; - else request.input.command = request.input.command.replace('"auto_decided":true', '"auto_decided":false'); - f.tools.push(request, result); expect(decision(f)).toBeNull(); -}); - -test('a literal command cannot treat a physical newline as argument whitespace', () => { - const f = clone(); - use(f, 'gstack-question-log').input.command = use(f, 'gstack-question-log').input.command.replace("gstack-question-log '", "gstack-question-log\n'"); - expect(decision(f)).toBeNull(); -}); - -for (const fallback of ['"LOGGED"', '" LOGGED "', '"\\x4cOGGED"', '-e "\\x4cOGGED"']) - test(`a failure branch cannot impersonate the question-log success marker: ${fallback}`, () => { - const f = clone(), request = use(f, 'gstack-question-log'); - request.input.command = request.input.command.replace('"log unavailable (best-effort)"', fallback); - expect(decision(f)).toBeNull(); - }); - -import completedModeCapture from './fixtures/auto-decide-completed-mode-f359.json'; -{ -const copy=()=>structuredClone(completedModeCapture); -const check=(f:any)=>findNativeAutoDecision(f.transcript,f.tools,f.options); -const message=(f:any)=>f.transcript.assistantMessages.find((m:any)=>m.text.includes('Mode decision done:')); -const logUse=(f:any)=>f.tools.find((t:any)=>t.kind==='use'&&t.input?.command?.includes('gstack-question-log')); -test('actual owned public attempt fails original and completes mode-only with full acknowledged authority',()=>{ - const f=copy();const v=check(f);expect(v?.option).toBe('HOLD SCOPE');expect(v?.questionLogToolUseId).toBe(logUse(f).toolUseId); -}); -const mutations:Recordvoid>={ - 'unlogged':f=>{const id=logUse(f).toolUseId;f.tools=f.tools.filter((t:any)=>t.toolUseId!==id)}, - 'failed log':f=>{f.tools.find((t:any)=>t.kind==='result'&&t.toolUseId===logUse(f).toolUseId).isError=true}, - 'masked log failure':f=>{logUse(f).input.command=logUse(f).input.command.replace('&& echo','; echo')}, - 'wrong returned marker':f=>{f.tools.find((t:any)=>t.kind==='result'&&t.toolUseId===logUse(f).toolUseId).content='LOG_FAILED (best-effort)'}, - 'unmatched quote':f=>{logUse(f).input.command=logUse(f).input.command.replace('"LOGGED"','"LOGGED')}, - 'foreign session':f=>{f.options.sessionId='foreign'}, - 'wrong mode':f=>{message(f).text=message(f).text.replace('done: HOLD SCOPE','done: SCOPE EXPANSION')}, - 'unfinished':f=>{message(f).text=message(f).text.replace('Mode decision done:','Mode decision pending:')}, - 'late declaration':f=>{message(f).timestamp=new Date(f.options.now+1000).toISOString()}, - 'prior declaration':f=>{message(f).timestamp=new Date(f.options.commandStartedAt-1000).toISOString()}, - 'cancelled':f=>{message(f).text+='\n\nI cancel this decision.'}, - 'wrong later completed mode':f=>{message(f).text+='\n\nMode decision done: SCOPE EXPANSION'}, - 'quoted declaration':f=>{message(f).text='> '+message(f).text}, - 'hypothetical':f=>{message(f).text='Example:\n'+message(f).text}, - 'conditional':f=>{message(f).text=message(f).text.replace('done: HOLD SCOPE','done: HOLD SCOPE (if approved)')}, - 'native question surfaced':f=>{f.transcript.calls.push({sessionId:f.options.sessionId})}, - 'wrong logged mode':f=>{logUse(f).input.command=logUse(f).input.command.replace('"user_choice":"HOLD SCOPE"','"user_choice":"SCOPE EXPANSION"')}, -}; -for(const [name,mutate] of Object.entries(mutations))test(name,()=>{const f=copy();mutate(f);expect(check(f)).toBeNull()}); - -for(const completion of ['done','complete','completed']) { - test(`completed mode class ${completion}`,()=>{const f=copy();message(f).text=message(f).text.replace('decision done:','decision '+completion+':');expect(check(f)?.option).toBe('HOLD SCOPE')}); - test(`conflicting later completed mode ${completion}`,()=>{const f=copy();message(f).text+='\n\nMode decision '+completion+': SCOPE EXPANSION';expect(check(f)).toBeNull()}); - test(`unfinished completed mode ${completion}`,()=>{const f=copy();message(f).text=message(f).text.replace('done: HOLD SCOPE',completion+': HOLD SCOPE (pending approval)');expect(check(f)).toBeNull()}); -} -test('paired single-quoted success token retains exact shell ACK',()=>{const f=copy();logUse(f).input.command=logUse(f).input.command.replace('"LOGGED"',"'LOGGED'");expect(check(f)?.option).toBe('HOLD SCOPE')}); -test('unpaired single-quoted success token cannot authenticate log',()=>{const f=copy();logUse(f).input.command=logUse(f).input.command.replace('"LOGGED"',"'LOGGED");expect(check(f)).toBeNull()}); - -} - -import statusFixture from './fixtures/auto-decide-completed-mode-f359.json'; -{ -const fixture=statusFixture; -const fixed=findNativeAutoDecision; -const copy=()=>structuredClone(fixture) as any; -const message=(f:any)=>f.transcript.assistantMessages.find((m:any)=>m.text.includes('Mode decision done:')); -const check=(f:any)=>fixed(f.transcript,f.tools,f.options); -test('current pending status retracts the completed owned mode',()=>{const f=copy();message(f).text+='\n\nMode decision pending: HOLD SCOPE';expect(check(f)).toBeNull()}); -for(const status of ['pending','pending approval','unfinished','incomplete','cancelled','canceled','withdrawn','retracted','revoked','undecided','proposed','not selected','not decided','not yet complete','in progress','on hold','unknown']){ - test(`unfinished declaration ${status}`,()=>{const f=copy();message(f).text=message(f).text.replace('decision done:','decision '+status+':');expect(check(f)).toBeNull()}); - test(`later unfinished status ${status}`,()=>{const f=copy();message(f).text+='\n\nMode decision '+status+': HOLD SCOPE';expect(check(f)).toBeNull()}); - test(`quoted historical status ${status}`,()=>{const f=copy();message(f).text+='\n\n> Historical example:\n> Mode decision '+status+': HOLD SCOPE';expect(check(f)?.option).toBe('HOLD SCOPE')}); -} -for(const status of ['done','complete','completed']){ - test(`same current completed field ${status}`,()=>{const f=copy();message(f).text+='\n\nMode decision '+status+': HOLD SCOPE';expect(check(f)?.option).toBe('HOLD SCOPE')}); - test(`completed conflicting field ${status}`,()=>{const f=copy();message(f).text+='\n\nMode decision '+status+': SCOPE EXPANSION';expect(check(f)).toBeNull()}); -} -for(const status of ['unfinished','incomplete','pending approval','cancelled','not completed'])test(`unfinished value suffix ${status}`,()=>{const f=copy();message(f).text+='\n\nMode decision done: HOLD SCOPE ('+status+')';expect(check(f)).toBeNull()}); -for(const text of ['Historical example: Mode decision pending: HOLD SCOPE','```\nMode decision pending: HOLD SCOPE\n```','"Mode decision cancelled: HOLD SCOPE"'])test(`unasserted historical field ${text}`,()=>{const f=copy();message(f).text+='\n\n'+text;expect(check(f)?.option).toBe('HOLD SCOPE')}); -} diff --git a/test/auto-decide-target-identity.test.ts b/test/auto-decide-target-identity.test.ts deleted file mode 100644 index f9accfd45..000000000 --- a/test/auto-decide-target-identity.test.ts +++ /dev/null @@ -1,128 +0,0 @@ -import { expect, test } from 'bun:test'; -import { findNativeAutoDecision } from './helpers/native-auto-decide'; -import capture from './fixtures/auto-decide-target-361c.json'; -const clone = () => structuredClone(capture) as any; -const message = (f: any) => f.transcript.assistantMessages.at(-1); -const decide = (f: any) => findNativeAutoDecision(f.transcript, f.tools, f.options); - -test('actual quoted current title and completed owned audit produce the original mode decision', () => { - const f = clone(), result = decide(f); - expect(result?.option).toBe('HOLD SCOPE'); - expect(result?.annotation).toBe(message(f).text); - expect(result?.stateRecord).toEqual(f.options.stateEvidence.records[0]); - expect(result?.preambleToolUseId).toBe('toolu_01KbsH6ybJxbNozwbSXywVbb'); -}); - -const title = 'deterministic skill-list ordering'; -const modes = ['HOLD SCOPE', 'SCOPE EXPANSION', 'SELECTIVE EXPANSION', 'SCOPE REDUCTION']; -for (const mode of modes) for (const quote of [(s: string) => `"${s}"`, (s: string) => `“${s}”`, (s: string) => `\`${s}\``]) { - for (const wrapper of ['', ' draft', ' plan']) test(`${mode} quoted title agrees with one audit wrapper: ${quote(title)}${wrapper}`, () => { - const f = clone(); - Object.assign(f.options.stateEvidence.records[0], { user_choice: mode, recommended: mode, question_summary: `Select review mode for ${title}${wrapper}` }); - message(f).text = `Decision: ${mode} for ${quote(title)}.\n\nMode: ${mode}, auto-selected using the saved preference.`; - expect(decide(f)?.option).toBe(mode); - expect(decide(f)?.annotation).toBe(message(f).text); - }); -} -for (const [declared, recorded] of [ - [`"${title}" draft`, `"${title}"`], - [`"${title}" plan`, `${title} draft`], - [title, `${title} draft`], - [`${title} draft`, title], - ['"release plan"', 'release plan draft'], - ['"what if ordering"', 'what if ordering draft'], - ['"ordering v2. current"', '"ordering v2. current" draft'], -]) test(`exact title identity with syntactic wrapper: ${declared} / ${recorded}`, () => { - const f = clone(); f.options.stateEvidence.records[0].question_summary = `Select mode for ${recorded}`; - message(f).text = `Decision: HOLD SCOPE for ${declared}.`; - expect(decide(f)?.option).toBe('HOLD SCOPE'); -}); - -test('quoted target and mode labels remain case insensitive', () => { - const f = clone(); message(f).text = `decision: hold scope FOR "${title}".`; - expect(decide(f)?.option).toBe('HOLD SCOPE'); -}); - -const negatives: Array<[string, string]> = [ - ['"deterministic skill-list sorting"', `${title} draft`], - ['"skill-list ordering"', `${title} draft`], - [`"${title}-v2"`, `${title} draft`], - [`"${title} extra"`, `${title} draft`], - ['"release"', '"release draft"'], - ['"release draft"', '"release"'], - ['"release plan"', 'release draft'], - ['release plan', 'release draft'], - ['"release draft plan"', 'release plan'], - ['"release plan draft"', '"release plan"'], - ['"draft release"', 'release'], - ['""', 'draft'], - ['" "', 'plan'], - [`"${title}" or "foreign"`, `${title} draft`], - [`"${title}" and another plan`, `${title} draft`], - [`"${title}`, `${title} draft`], - [`${title}"`, `${title} draft`], - ['"future plan"', 'future plan'], - ['future', 'future plan'], - ['previous', 'previous draft'], - ['"previous draft"', 'previous draft'], - ['"another draft"', 'another draft'], - ['"next plan"', 'next plan'], -]; -for (const [declared, recorded] of negatives) { - test(`target cannot borrow a named or historical match: ${declared} / ${recorded}`, () => { - const f = clone(); f.options.stateEvidence.records[0].question_summary = `Select mode for ${recorded}`; - message(f).text = `Decision: HOLD SCOPE for ${declared}.`; - expect(decide(f)).toBeNull(); - }); - test(`later agreeing Mode does not erase invalid target: ${declared} / ${recorded}`, () => { - const f = clone(); f.options.stateEvidence.records[0].question_summary = `Select mode for ${recorded}`; - message(f).text = `Decision: HOLD SCOPE for ${declared}.\n\nMode: HOLD SCOPE, auto-selected.`; - expect(decide(f)).toBeNull(); - }); -} - -for (const wrap of [ - (s: string) => `"${s}"`, (s: string) => `“${s}”`, (s: string) => `\`${s}\``, - (s: string) => `> ${s}`, (s: string) => ` ${s}`, (s: string) => `\`\`\`text\n${s}\n\`\`\``, - (s: string) => `Example:\n${s}`, (s: string) => `Previous review:\n${s}`, -]) test(`only an asserted field can own a quoted target: ${wrap('Decision')}`, () => { - const f = clone(); message(f).text = wrap(`Decision: HOLD SCOPE for "${title}".`); - expect(decide(f)).toBeNull(); -}); - -for (const value of [ - `HOLD SCOPE for "${title}" if approved`, `HOLD SCOPE for "${title}", pending approval`, - `not HOLD SCOPE for "${title}"`, `HOLD SCOPE for "${title}"; SCOPE EXPANSION`, - `HOLD SCOPE for "${title}" (withdrawn)`, `HOLD SCOPE for "${title}" (I will select it)`, -]) test(`quoted name cannot hide a lifecycle veto: ${value}`, () => { - const f = clone(); message(f).text = `Decision: ${value}.\n\nMode: HOLD SCOPE.`; - expect(decide(f)).toBeNull(); -}); -for (const suffix of [ - '\n\nCorrection: Mode: SCOPE EXPANSION.', - '\n\nCorrection: I withdraw this decision.', - `\n\nDecision: HOLD SCOPE for "foreign target".`, - '\n\nMode pending: HOLD SCOPE.', -]) test(`a later contradiction remains effective: ${suffix}`, () => { - const f = clone(); message(f).text += suffix; expect(decide(f)).toBeNull(); -}); -for (const [name, mutate] of Object.entries({ - 'missing owned log': (f: any) => { f.options.stateEvidence.records = []; }, - 'duplicate owned log': (f: any) => { f.options.stateEvidence.records.push({ ...f.options.stateEvidence.records[0] }); }, - 'foreign audit session': (f: any) => { f.options.stateEvidence.records[0].session_id = 'foreign'; }, - 'different audit choice': (f: any) => { f.options.stateEvidence.records[0].user_choice = 'SCOPE EXPANSION'; }, - 'wrong preference': (f: any) => { f.options.stateEvidence.preference = 'ask'; }, - 'missing preamble ACK': (f: any) => { f.tools = f.tools.filter((e: any) => !(e.kind === 'result' && e.toolUseId === 'toolu_01KbsH6ybJxbNozwbSXywVbb')); }, - 'native question': (f: any) => { f.transcript.calls.push({sessionId:f.options.sessionId}); }, - 'prose question': (f: any) => { f.options.proseQuestionObserved = true; }, - 'decision before log': (f: any) => { message(f).timestamp = new Date(Date.parse(f.options.stateEvidence.records[0].ts) - 1).toISOString(); }, - 'wrong native session': (f: any) => { f.options.sessionId = 'foreign'; }, - 'future log': (f: any) => { f.options.stateEvidence.records[0].ts = new Date(f.options.now + 1).toISOString(); }, -})) test(`actual quoted target retains ${name} boundary`, () => { - const f = clone(); mutate(f); expect(decide(f)).toBeNull(); -}); -for (const preposition of ['for', 'FOR']) test(`a quoted lifecycle word belongs to its title with ${preposition}`, () => { - const f = clone(); f.options.stateEvidence.records[0].question_summary = 'Select mode for Pending notifications draft'; - message(f).text = `Decision: HOLD SCOPE ${preposition} "Pending notifications".`; - expect(decide(f)?.option).toBe('HOLD SCOPE'); -}); diff --git a/test/auto-decision-state.test.ts b/test/auto-decision-state.test.ts deleted file mode 100644 index d973c0966..000000000 --- a/test/auto-decision-state.test.ts +++ /dev/null @@ -1,87 +0,0 @@ -import { expect, test } from 'bun:test'; -import * as fs from 'node:fs'; -import * as os from 'node:os'; -import * as path from 'node:path'; -import { bindAutoDecisionState } from './helpers/auto-decision-state'; -import { findNativeAutoDecision } from './helpers/native-auto-decide'; -import capture from './fixtures/auto-decide-state-cab3.json'; - -const clone = () => structuredClone(capture) as any; -const qid = 'plan-ceo-review-mode'; -function state(f: any) { - const use = f.tools.find((e: any) => e.input?.command?.includes('gstack-question-log')); - // Synthetic file witness, built from the actual literal request. The original - // run did not retain this file, and is still a failed paid attempt. - const record = JSON.parse(/gstack-question-log '(\{[^\n]*\})'/.exec(use.input.command)![1]!); - record.source = 'agent'; - record.ts = f.tools.find((e: any) => e.kind === 'result' && e.toolUseId === use.toolUseId).timestamp; - return { questionId: qid, preference: 'never-ask' as const, records: [record] }; -} -const decide = (f: any) => findNativeAutoDecision(f.transcript, f.tools, f.options); -const mode = (f: any) => f.transcript.assistantMessages.find((m: any) => m.text.startsWith('**Mode:')); - -test('original captured retry cannot prove a masked log succeeded', () => { - expect(decide(clone())).toBeNull(); -}); -test('actual retry declaration plus a completed owned append proves the chosen mode', () => { - const f = clone(); f.options.stateEvidence = state(f); - const result = decide(f); - expect(result?.option).toBe('HOLD SCOPE'); - expect(result?.stateRecord).toEqual(f.options.stateEvidence.records[0]); - expect(result?.questionLogToolUseId).toBeUndefined(); -}); - -for (const [name, mutate] of Object.entries({ - 'foreign record session': (f: any) => { f.options.stateEvidence.records[0].session_id = 'foreign'; }, - 'wrong question': (f: any) => { f.options.stateEvidence.questionId = 'wrong'; }, - 'wrong skill': (f: any) => { f.options.stateEvidence.records[0].skill = 'plan-eng-review'; }, - 'nonautomatic record': (f: any) => { f.options.stateEvidence.records[0].auto_decided = false; }, - 'string flag': (f: any) => { f.options.stateEvidence.records[0].auto_decided = 'true'; }, - 'wrong source': (f: any) => { f.options.stateEvidence.records[0].source = 'hook'; }, - 'different preference': (f: any) => { f.options.stateEvidence.preference = 'always-ask'; }, - 'missing append': (f: any) => { f.options.stateEvidence.records = []; }, - 'duplicate append': (f: any) => { f.options.stateEvidence.records.push({ ...f.options.stateEvidence.records[0] }); }, - 'contradictory recommendation': (f: any) => { f.options.stateEvidence.records[0].recommended = 'SCOPE EXPANSION'; }, - 'empty summary': (f: any) => { f.options.stateEvidence.records[0].question_summary = ''; }, - 'old record': (f: any) => { f.options.stateEvidence.records[0].ts = new Date(f.options.commandStartedAt - 1).toISOString(); }, - 'future record': (f: any) => { f.options.stateEvidence.records[0].ts = new Date(f.options.now + 1).toISOString(); }, - 'record after declaration': (f: any) => { f.options.stateEvidence.records[0].ts = new Date(Date.parse(mode(f).timestamp) + 1).toISOString(); }, - 'invalid timestamp': (f: any) => { f.options.stateEvidence.records[0].ts = 'invalid'; }, - 'actual native question': (f: any) => { f.transcript.calls.push({ sessionId: f.options.sessionId }); }, - 'actual prose question': (f: any) => { f.options.proseQuestionObserved = true; }, - 'failed preamble': (f: any) => { f.tools.find((e: any) => e.kind === 'result' && e.content?.includes('SKILL_START_PROTO')).isError = true; }, - 'quoted declaration': (f: any) => { mode(f).text = '> Mode: HOLD SCOPE (saved preference).'; }, - 'conditional declaration': (f: any) => { mode(f).text = 'Mode: HOLD SCOPE (if approved).'; }, - 'later withdrawal': (f: any) => { mode(f).text += '\n\nCorrection: I withdraw this decision.'; }, - 'later different mode': (f: any) => { mode(f).text += '\n\nMode: SCOPE EXPANSION (saved preference).'; }, -})) test(`owned log witness rejects ${name}`, () => { - const f = clone(); f.options.stateEvidence = state(f); mutate(f); expect(decide(f)).toBeNull(); -}); - -function withState(check: (x: { root: string; project: string; pref: string; log: string; bind: () => ReturnType }) => void) { - const root = fs.realpathSync(fs.mkdtempSync(path.join(os.tmpdir(), 'auto-state-'))); - const project = path.join(root, 'projects', 'fixture'); fs.mkdirSync(project, { recursive: true }); - const pref = path.join(project, 'question-preferences.json'), log = path.join(project, 'question-log.jsonl'); - fs.writeFileSync(pref, JSON.stringify({ [qid]: 'never-ask' })); - const bind = () => bindAutoDecisionState({ stateRoot: root, projectSlug: 'fixture' }, { GSTACK_STATE_ROOT: root }, 'plan-ceo-review'); - try { check({ root, project, pref, log, bind }); } finally { fs.rmSync(root, { recursive: true, force: true }); } -} -test('state witness binds before launch and observes only completed owned file contents', () => withState(({ log, bind }) => { - const read = bind(); expect(read()).toBeUndefined(); - const record = state(clone()).records[0]; fs.writeFileSync(log, JSON.stringify(record) + '\n'); - expect(read()?.records).toEqual([record]); -})); -for (const scenario of ['existing-log', 'preference-change', 'malformed-log', 'log-symlink', 'preference-symlink', 'wrong-root', 'path-escape']) - test(`state binding rejects ${scenario}`, () => withState(({ root, pref, log, bind }) => { - if (scenario === 'wrong-root' || scenario === 'path-escape') { - expect(() => bindAutoDecisionState({ stateRoot: root, projectSlug: scenario === 'path-escape' ? '../fixture' : 'fixture' }, - { GSTACK_STATE_ROOT: scenario === 'wrong-root' ? root + '-other' : root }, 'plan-ceo-review')).toThrow(); return; - } - if (scenario === 'existing-log') { fs.writeFileSync(log, '{}\n'); expect(bind).toThrow('fresh attempt'); return; } - const read = bind(); - if (scenario === 'preference-change') fs.writeFileSync(pref, JSON.stringify({ [qid]: 'always-ask' })); - if (scenario === 'malformed-log') fs.writeFileSync(log, '{'); - if (scenario === 'log-symlink') fs.symlinkSync(pref, log); - if (scenario === 'preference-symlink') { fs.renameSync(pref, pref + '.real'); fs.symlinkSync(pref + '.real', pref); } - expect(read()).toBeUndefined(); - })); diff --git a/test/autoplan-phase-dash-ao.test.ts b/test/autoplan-phase-dash-ao.test.ts deleted file mode 100644 index b053112a1..000000000 --- a/test/autoplan-phase-dash-ao.test.ts +++ /dev/null @@ -1,78 +0,0 @@ -import { expect, test } from 'bun:test'; -import fixture from './fixtures/autoplan-phase-dash-ao.json'; -import { autoplanPhaseCompletions } from './helpers/autoplan-phase-observer'; -import { E2E_TOUCHFILES, selectTests } from './helpers/touchfiles'; -import type { PlanCountTranscript } from './helpers/plan-count-transcript'; - -const at = Date.parse(fixture.message.timestamp); -const transcript = (text = fixture.message.text): PlanCountTranscript => ({ - status: 'ready', calls: [], assistantMessages: [{ ...fixture.message, text }], -}); -const hits = (text: string) => autoplanPhaseCompletions(transcript(text), at - 1); - -test('exact owned DX dash declaration adds only DX at its native timestamp', () => { - expect(hits(fixture.message.text)).toEqual([{ phase: 2.5, ts: at }]); - const all = autoplanPhaseCompletions({ status: 'ready', calls: [], - assistantMessages: fixture.orderedMessages }, fixture.commandLowerBound); - expect(all).toEqual([...fixture.actualHits, { phase: 2.5, ts: at }]); - expect(all.map(hit => hit.phase)).toEqual([1, 2, 2.5]); -}); - -test('em and en dash spacing share the existing completed declaration forms', () => { - for (const dash of ['—', '–']) for (const before of ['', ' ']) for (const after of ['', ' ']) { - expect(hits(fixture.message.text.replace('complete—', `complete${before}${dash}${after}`))) - .toEqual([{ phase: 2.5, ts: at }]); - for (const phase of [1, 2, 2.5, 3]) for (const state of ['complete', 'completed', 'done', 'finished', 'wrapped up']) { - expect(hits(`Phase ${phase} is ${state}${before}${dash}${after}Work retained.`)) - .toEqual([{ phase, ts: at }]); - } - } - expect(hits('**Phase 2.5 complete** — Work retained.')).toEqual([{ phase: 2.5, ts: at }]); -}); - -test('dash continuations cannot turn a conditional, quotation, question or denial into completion', () => { - for (const dash of ['—', '–']) for (const tail of [ - '', 'if approved.', 'unless the checks fail.', 'when review finishes.', - 'once the reviewer signs off.', 'pending final checks.', 'maybe tomorrow.', - 'perhaps it is complete.', 'would be complete after review.', - 'not complete yet.', 'the phase is not complete.', 'this completion is withdrawn.', - 'actually never finished.', 'this completion is superseded.', - 'provided the remaining checks pass.', 'this completion is rejected.', - 'the completion announcement is retracted.', 'actually incomplete.', - 'the review remains pending.', 'Work retained?', 'is this complete?', - 'Source excerpt: Work retained.', 'Earlier review: Work retained.', - 'the historical example says work is retained.', '"Work retained."', - ]) expect(hits(`Phase 2.5 complete ${dash} ${tail}`), tail).toEqual([]); - for (const text of [ - 'If approved, Phase 2.5 complete—Work retained.', - 'Phase 2.5 is not complete—Work retained.', - 'Phase 2.5 complete?—Work retained.', - '> Phase 2.5 complete—Work retained.', - '"Phase 2.5 complete—Work retained."', - 'Source excerpt:\nPhase 2.5 complete—Work retained.', - 'Example:\nPhase 2.5 complete—Work retained.\nPhase 3 complete—Work retained.', - '```text\nPhase 2.5 complete—Work retained.\n```', - ' Phase 2.5 complete—Work retained.', - '# Phase 2.5 complete—Work retained.', - 'Phase 2.5 (Eng review) complete—Work retained.', - ]) expect(hits(text), text).toEqual([]); -}); - -test('dash support keeps ready/current native evidence and first-hit ordering', () => { - for (const status of ['missing', 'error'] as const) { - expect(autoplanPhaseCompletions({ ...transcript(), status }, at - 1)).toEqual([]); - } - expect(autoplanPhaseCompletions(transcript(), at + 1)).toEqual([]); - expect(autoplanPhaseCompletions({ ...transcript(), assistantMessages: [ - { ...fixture.message, timestamp: 'invalid' }, - ] }, at - 1)).toEqual([]); - const later = { ...fixture.message, timestamp: new Date(at + 1).toISOString() }; - expect(autoplanPhaseCompletions({ ...transcript(), assistantMessages: [later, fixture.message] }, at - 1)) - .toEqual([{ phase: 2.5, ts: at }]); -}); - -test('dash fixture and regression select only the existing AP owner', () => { - for (const file of ['test/autoplan-phase-dash-ao.test.ts', 'test/fixtures/autoplan-phase-dash-ao.json']) { - expect(selectTests([file], E2E_TOUCHFILES).selected).toEqual([]); - } -}); diff --git a/test/autoplan-phase-observer.test.ts b/test/autoplan-phase-observer.test.ts index 5e9730fe4..0ccb558d1 100644 --- a/test/autoplan-phase-observer.test.ts +++ b/test/autoplan-phase-observer.test.ts @@ -6,6 +6,8 @@ import { pathToFileURL } from 'node:url'; import { autoplanPhaseCompletions } from './helpers/autoplan-phase-observer'; import type { PlanCountTranscript } from './helpers/plan-count-transcript'; import { E2E_TOUCHFILES, selectTests } from './helpers/touchfiles'; +import fixture_autoplan_phase_dash_ao from './fixtures/autoplan-phase-dash-ao.json'; +import actual_autoplan_with_result_au from './fixtures/autoplan-with-result-au.json'; const START = Date.parse('2026-09-08T16:00:00.000Z'); const transcript = (...messages: Array<[number, string]>): PlanCountTranscript => ({ @@ -298,3 +300,191 @@ try { } }, 15_000); }); + +describe('autoplan-phase-dash-ao', () => { +const fixture = fixture_autoplan_phase_dash_ao; +const at = Date.parse(fixture.message.timestamp); +const transcript = (text = fixture.message.text): PlanCountTranscript => ({ + status: 'ready', calls: [], assistantMessages: [{ ...fixture.message, text }], +}); +const hits = (text: string) => autoplanPhaseCompletions(transcript(text), at - 1); + +test('exact owned DX dash declaration adds only DX at its native timestamp', () => { + expect(hits(fixture.message.text)).toEqual([{ phase: 2.5, ts: at }]); + const all = autoplanPhaseCompletions({ status: 'ready', calls: [], + assistantMessages: fixture.orderedMessages }, fixture.commandLowerBound); + expect(all).toEqual([...fixture.actualHits, { phase: 2.5, ts: at }]); + expect(all.map(hit => hit.phase)).toEqual([1, 2, 2.5]); +}); + +test('em and en dash spacing share the existing completed declaration forms', () => { + for (const dash of ['—', '–']) for (const before of ['', ' ']) for (const after of ['', ' ']) { + expect(hits(fixture.message.text.replace('complete—', `complete${before}${dash}${after}`))) + .toEqual([{ phase: 2.5, ts: at }]); + for (const phase of [1, 2, 2.5, 3]) for (const state of ['complete', 'completed', 'done', 'finished', 'wrapped up']) { + expect(hits(`Phase ${phase} is ${state}${before}${dash}${after}Work retained.`)) + .toEqual([{ phase, ts: at }]); + } + } + expect(hits('**Phase 2.5 complete** — Work retained.')).toEqual([{ phase: 2.5, ts: at }]); +}); + +test('dash continuations cannot turn a conditional, quotation, question or denial into completion', () => { + for (const dash of ['—', '–']) for (const tail of [ + '', 'if approved.', 'unless the checks fail.', 'when review finishes.', + 'once the reviewer signs off.', 'pending final checks.', 'maybe tomorrow.', + 'perhaps it is complete.', 'would be complete after review.', + 'not complete yet.', 'the phase is not complete.', 'this completion is withdrawn.', + 'actually never finished.', 'this completion is superseded.', + 'provided the remaining checks pass.', 'this completion is rejected.', + 'the completion announcement is retracted.', 'actually incomplete.', + 'the review remains pending.', 'Work retained?', 'is this complete?', + 'Source excerpt: Work retained.', 'Earlier review: Work retained.', + 'the historical example says work is retained.', '"Work retained."', + ]) expect(hits(`Phase 2.5 complete ${dash} ${tail}`), tail).toEqual([]); + for (const text of [ + 'If approved, Phase 2.5 complete—Work retained.', + 'Phase 2.5 is not complete—Work retained.', + 'Phase 2.5 complete?—Work retained.', + '> Phase 2.5 complete—Work retained.', + '"Phase 2.5 complete—Work retained."', + 'Source excerpt:\nPhase 2.5 complete—Work retained.', + 'Example:\nPhase 2.5 complete—Work retained.\nPhase 3 complete—Work retained.', + '```text\nPhase 2.5 complete—Work retained.\n```', + ' Phase 2.5 complete—Work retained.', + '# Phase 2.5 complete—Work retained.', + 'Phase 2.5 (Eng review) complete—Work retained.', + ]) expect(hits(text), text).toEqual([]); +}); + +test('dash support keeps ready/current native evidence and first-hit ordering', () => { + for (const status of ['missing', 'error'] as const) { + expect(autoplanPhaseCompletions({ ...transcript(), status }, at - 1)).toEqual([]); + } + expect(autoplanPhaseCompletions(transcript(), at + 1)).toEqual([]); + expect(autoplanPhaseCompletions({ ...transcript(), assistantMessages: [ + { ...fixture.message, timestamp: 'invalid' }, + ] }, at - 1)).toEqual([]); + const later = { ...fixture.message, timestamp: new Date(at + 1).toISOString() }; + expect(autoplanPhaseCompletions({ ...transcript(), assistantMessages: [later, fixture.message] }, at - 1)) + .toEqual([{ phase: 2.5, ts: at }]); +}); +}); + +describe('autoplan-with-result-au', () => { +const actual = actual_autoplan_with_result_au; +const at=Date.parse(actual.timestamp); +const transcript=(text=actual.text):PlanCountTranscript=>({status:'ready',calls:[],assistantMessages:[{...actual,text}]}); +const observe=(text:string)=>autoplanPhaseCompletions(transcript(text),at-1); + +test('the exact first AU DX completion retains its native timestamp without crediting the Eng transition',()=>{ + expect(autoplanPhaseCompletions(transcript(),at-1)).toEqual([{phase:2.5,ts:at}]); + expect(actual.sessionId).toBe('78ce9c42-e5f7-4595-81ea-7d9bb8b4345c'); + expect(actual.timestamp).toBe('2026-09-10T21:35:46.209Z'); +}); + +test('affirmative result clauses share phase identity and the existing completion vocabulary',()=>{ + for(const [phase,name] of [[1,'CEO'],[2,'Design review'],[2.5,'DX'],[3,'Engineering review']] as const) + for(const state of ['complete','completed','done','finished','wrapped up']) + for(const result of ['22 findings recorded in the plan.','the score at 8/10.','all adopted changes written; moving to the next phase.']) { + expect(observe(`Phase ${phase} (${name}) is ${state} with ${result}`)).toEqual([{phase,ts:at}]); + } + expect(observe('**Phase 2.5 wrapped up** with 22 findings retained.')).toEqual([{phase:2.5,ts:at}]); +}); + +const rejected=[ + 'Phase 2.5 wrapped up with ', + 'Phase 2.5 wrapped up without the review.', + 'Phase 2.5 will be complete with 22 findings.', + 'Phase 2.5 is not complete with 22 findings.', + 'Phase 2.5 complete with no completed review.', + 'Phase 2.5 complete with findings still pending.', + 'Phase 2.5 complete with 22 findings if the review finishes.', + 'Phase 2.5 complete with 22 findings once approved.', + 'Phase 2.5 complete with 22 findings when the review ends.', + 'Phase 2.5 complete with 22 findings unless the review fails.', + 'Phase 2.5 complete with 22 findings provided the reviewer agrees.', + 'Phase 2.5 complete with 22 findings?','Phase 2.5 complete with results that will arrive tomorrow.', + 'Phase 2.5 complete with maybe 22 findings.','Phase 2.5 complete with an unfinished review.', + 'Phase 2.5 complete with 22 findings. This phase is withdrawn.', + 'Phase 2.5 complete with 22 findings. This phase is "withdrawn".', + 'Phase 2.5 complete with 22 findings. This phase is not complete.', + 'Phase 2.5 complete with 22 findings. This phase is retracted.', + 'Phase 2.5 complete with 22 findings. The declaration is superseded.', + 'Phase 2.5 complete with a historical example.', + 'Phase 2.5 complete with source instructions.', + 'Phase 2.5 complete with "22 findings recorded".', + 'Phase 2.5 complete with \'22 findings recorded\'.', + 'Phase 2.5 (Design) complete with 22 findings.', + 'Phase 2.5 (DX review if approved) complete with 22 findings.', + 'Phase 4 complete with 22 findings.','Phase 2.1 complete with 22 findings.', + '# Phase 2.5 complete with 22 findings.', + '> Phase 2.5 complete with 22 findings.', + '"Phase 2.5 complete with 22 findings."', + '- Phase 2.5 complete with 22 findings.', + '| Phase 2.5 complete with 22 findings. |', + ' Phase 2.5 complete with 22 findings.', + '\tPhase 2.5 complete with 22 findings.', + '```text\nPhase 2.5 complete with 22 findings.\n```', + '~~~text\nPhase 2.5 complete with 22 findings.\n~~~', + 'Source:\nPhase 2.5 complete with 22 findings.', + 'Historical example:\nPhase 2.5 complete with 22 findings.', + 'Historical review:\nPhase 2.5 complete with 22 findings.', + '**Historical review:**\nPhase 2.5 complete with 22 findings.', + '**Source:**\nPhase 2.5 complete with 22 findings.', + 'Hypothetical scenario:\nPhase 2.5 complete with 22 findings.', + 'Phase 2.5 complete with 22 findings.\n```text\nexample text\n````\nThis phase is withdrawn.', + 'Earlier review:\nPhase 2.5 complete with 22 findings.', + 'Phase 2.5 complete with a hypothetical 8/10 score.', + 'Phase 2.5 complete with 22 findings.\nThis phase is withdrawn.', + 'Phase 2.5 complete with 22 findings.\nThis phase is \"withdrawn\".', + 'Phase 2.5 complete with 22 findings.\n**Phase 2.5** is ‘withdrawn’.', + 'Phase 2.5 complete with 22 findings.\nCurrent status: this phase is no longer current.', + 'The template says:\n\nPhase 2.5 complete with 22 findings.', + 'Example:\nPhase 1 complete with findings.\nPhase 2.5 complete with findings.', +]; +test.each(rejected)('%s cannot supply completion',text=>expect(observe(text)).toEqual([])); + +test('quoted summaries retain their existing concrete-consensus requirement',()=>{ + const summary='> Phase 2.5 complete with 22 findings retained.\n> Consensus: 22/22 accepted.\n> Moving to Phase 3.'; + expect(observe(summary)).toEqual([{phase:2.5,ts:at}]); + for(const text of [summary.replace('22/22','[N]/22'),'Example:\n'+summary,summary.replace('22/22','X/Y')]) + expect(observe(text)).toEqual([]); +}); + +test('native readiness, timestamp, duplicate and observed-order rules remain intact',()=>{ + for(const status of ['missing','error'] as const) + expect(autoplanPhaseCompletions({...transcript(),status},at-1)).toEqual([]); + expect(autoplanPhaseCompletions(transcript(),at+1)).toEqual([]); + expect(autoplanPhaseCompletions({...transcript(),assistantMessages:[{...actual,timestamp:'invalid'}]},at-1)).toEqual([]); + const data=transcript();data.assistantMessages.push({...actual,timestamp:new Date(at+1).toISOString()}); + expect(autoplanPhaseCompletions(data,at-1)).toEqual([{phase:2.5,ts:at}]); + data.assistantMessages.unshift({...actual,text:'Phase 3 complete with 7 findings retained.',timestamp:new Date(at-10).toISOString()}); + expect(autoplanPhaseCompletions(data,at-11)).toEqual([{phase:3,ts:at-10},{phase:2.5,ts:at}]); +}); +test('quoted history and a foreign phase withdrawal do not cancel the current completed result',()=>{ + for(const suffix of [ + '> This phase is withdrawn.', + 'Historical note: "This phase is withdrawn."', + 'Example:\nThis phase is withdrawn.', + '```text\nThis phase is withdrawn.\n```', + 'Phase 2 is withdrawn.', + 'Phase 3 complete.\nThis phase is withdrawn.', + ]) expect(observe('Phase 2.5 complete with 22 findings retained.\n'+suffix).some(hit=>hit.phase===2.5)).toBe(true); + expect(observe('Phase 2.5 complete with 22 findings.\nHistorical note:\nThis phase is withdrawn.\nCurrent status: Phase 2.5 is withdrawn.')).toEqual([]); +}); + + +test('a current Markdown status heading resets historical context for an owned withdrawal',()=>{ + const prefix='Phase 2.5 complete with 22 findings retained.\nHistorical note:\nThis phase is withdrawn.\n'; + expect(observe(prefix+'## Current status\nPhase 2.5 is withdrawn.')).toEqual([]); + expect(observe(prefix+'`## Current status`\nThis phase is withdrawn.')).toEqual([{phase:2.5,ts:at}]); +}); + +test('inline code around an owned status is scalar formatting while a whole quoted statement stays literal',()=>{ + const prefix='Phase 2.5 complete with 22 findings retained.\n'; + expect(observe(prefix+'This phase is `withdrawn`.')).toEqual([]); + for(const literal of ['`This phase is withdrawn.`','"This phase is withdrawn."','```text\nThis phase is withdrawn.\n```']) + expect(observe(prefix+literal)).toEqual([{phase:2.5,ts:at}]); +}); +}); diff --git a/test/autoplan-with-result-au.test.ts b/test/autoplan-with-result-au.test.ts deleted file mode 100644 index 82b36c8a6..000000000 --- a/test/autoplan-with-result-au.test.ts +++ /dev/null @@ -1,126 +0,0 @@ -import {expect, test} from 'bun:test'; -import actual from './fixtures/autoplan-with-result-au.json'; -import {autoplanPhaseCompletions} from './helpers/autoplan-phase-observer'; -import {E2E_TOUCHFILES, selectTests} from './helpers/touchfiles'; -import type {PlanCountTranscript} from './helpers/plan-count-transcript'; -const at=Date.parse(actual.timestamp); -const transcript=(text=actual.text):PlanCountTranscript=>({status:'ready',calls:[],assistantMessages:[{...actual,text}]}); -const observe=(text:string)=>autoplanPhaseCompletions(transcript(text),at-1); - -test('the exact first AU DX completion retains its native timestamp without crediting the Eng transition',()=>{ - expect(autoplanPhaseCompletions(transcript(),at-1)).toEqual([{phase:2.5,ts:at}]); - expect(actual.sessionId).toBe('78ce9c42-e5f7-4595-81ea-7d9bb8b4345c'); - expect(actual.timestamp).toBe('2026-09-10T21:35:46.209Z'); -}); - -test('affirmative result clauses share phase identity and the existing completion vocabulary',()=>{ - for(const [phase,name] of [[1,'CEO'],[2,'Design review'],[2.5,'DX'],[3,'Engineering review']] as const) - for(const state of ['complete','completed','done','finished','wrapped up']) - for(const result of ['22 findings recorded in the plan.','the score at 8/10.','all adopted changes written; moving to the next phase.']) { - expect(observe(`Phase ${phase} (${name}) is ${state} with ${result}`)).toEqual([{phase,ts:at}]); - } - expect(observe('**Phase 2.5 wrapped up** with 22 findings retained.')).toEqual([{phase:2.5,ts:at}]); -}); - -const rejected=[ - 'Phase 2.5 wrapped up with ', - 'Phase 2.5 wrapped up without the review.', - 'Phase 2.5 will be complete with 22 findings.', - 'Phase 2.5 is not complete with 22 findings.', - 'Phase 2.5 complete with no completed review.', - 'Phase 2.5 complete with findings still pending.', - 'Phase 2.5 complete with 22 findings if the review finishes.', - 'Phase 2.5 complete with 22 findings once approved.', - 'Phase 2.5 complete with 22 findings when the review ends.', - 'Phase 2.5 complete with 22 findings unless the review fails.', - 'Phase 2.5 complete with 22 findings provided the reviewer agrees.', - 'Phase 2.5 complete with 22 findings?','Phase 2.5 complete with results that will arrive tomorrow.', - 'Phase 2.5 complete with maybe 22 findings.','Phase 2.5 complete with an unfinished review.', - 'Phase 2.5 complete with 22 findings. This phase is withdrawn.', - 'Phase 2.5 complete with 22 findings. This phase is "withdrawn".', - 'Phase 2.5 complete with 22 findings. This phase is not complete.', - 'Phase 2.5 complete with 22 findings. This phase is retracted.', - 'Phase 2.5 complete with 22 findings. The declaration is superseded.', - 'Phase 2.5 complete with a historical example.', - 'Phase 2.5 complete with source instructions.', - 'Phase 2.5 complete with "22 findings recorded".', - 'Phase 2.5 complete with \'22 findings recorded\'.', - 'Phase 2.5 (Design) complete with 22 findings.', - 'Phase 2.5 (DX review if approved) complete with 22 findings.', - 'Phase 4 complete with 22 findings.','Phase 2.1 complete with 22 findings.', - '# Phase 2.5 complete with 22 findings.', - '> Phase 2.5 complete with 22 findings.', - '"Phase 2.5 complete with 22 findings."', - '- Phase 2.5 complete with 22 findings.', - '| Phase 2.5 complete with 22 findings. |', - ' Phase 2.5 complete with 22 findings.', - '\tPhase 2.5 complete with 22 findings.', - '```text\nPhase 2.5 complete with 22 findings.\n```', - '~~~text\nPhase 2.5 complete with 22 findings.\n~~~', - 'Source:\nPhase 2.5 complete with 22 findings.', - 'Historical example:\nPhase 2.5 complete with 22 findings.', - 'Historical review:\nPhase 2.5 complete with 22 findings.', - '**Historical review:**\nPhase 2.5 complete with 22 findings.', - '**Source:**\nPhase 2.5 complete with 22 findings.', - 'Hypothetical scenario:\nPhase 2.5 complete with 22 findings.', - 'Phase 2.5 complete with 22 findings.\n```text\nexample text\n````\nThis phase is withdrawn.', - 'Earlier review:\nPhase 2.5 complete with 22 findings.', - 'Phase 2.5 complete with a hypothetical 8/10 score.', - 'Phase 2.5 complete with 22 findings.\nThis phase is withdrawn.', - 'Phase 2.5 complete with 22 findings.\nThis phase is \"withdrawn\".', - 'Phase 2.5 complete with 22 findings.\n**Phase 2.5** is ‘withdrawn’.', - 'Phase 2.5 complete with 22 findings.\nCurrent status: this phase is no longer current.', - 'The template says:\n\nPhase 2.5 complete with 22 findings.', - 'Example:\nPhase 1 complete with findings.\nPhase 2.5 complete with findings.', -]; -test.each(rejected)('%s cannot supply completion',text=>expect(observe(text)).toEqual([])); - -test('quoted summaries retain their existing concrete-consensus requirement',()=>{ - const summary='> Phase 2.5 complete with 22 findings retained.\n> Consensus: 22/22 accepted.\n> Moving to Phase 3.'; - expect(observe(summary)).toEqual([{phase:2.5,ts:at}]); - for(const text of [summary.replace('22/22','[N]/22'),'Example:\n'+summary,summary.replace('22/22','X/Y')]) - expect(observe(text)).toEqual([]); -}); - -test('native readiness, timestamp, duplicate and observed-order rules remain intact',()=>{ - for(const status of ['missing','error'] as const) - expect(autoplanPhaseCompletions({...transcript(),status},at-1)).toEqual([]); - expect(autoplanPhaseCompletions(transcript(),at+1)).toEqual([]); - expect(autoplanPhaseCompletions({...transcript(),assistantMessages:[{...actual,timestamp:'invalid'}]},at-1)).toEqual([]); - const data=transcript();data.assistantMessages.push({...actual,timestamp:new Date(at+1).toISOString()}); - expect(autoplanPhaseCompletions(data,at-1)).toEqual([{phase:2.5,ts:at}]); - data.assistantMessages.unshift({...actual,text:'Phase 3 complete with 7 findings retained.',timestamp:new Date(at-10).toISOString()}); - expect(autoplanPhaseCompletions(data,at-11)).toEqual([{phase:3,ts:at-10},{phase:2.5,ts:at}]); -}); - -test('the regression and exact public message select only the existing Autoplan workflow',()=>{ - for(const file of ['test/autoplan-with-result-au.test.ts','test/fixtures/autoplan-with-result-au.json']) - expect(selectTests([file],E2E_TOUCHFILES,[]).selected).toEqual([]); -}); - - -test('quoted history and a foreign phase withdrawal do not cancel the current completed result',()=>{ - for(const suffix of [ - '> This phase is withdrawn.', - 'Historical note: "This phase is withdrawn."', - 'Example:\nThis phase is withdrawn.', - '```text\nThis phase is withdrawn.\n```', - 'Phase 2 is withdrawn.', - 'Phase 3 complete.\nThis phase is withdrawn.', - ]) expect(observe('Phase 2.5 complete with 22 findings retained.\n'+suffix).some(hit=>hit.phase===2.5)).toBe(true); - expect(observe('Phase 2.5 complete with 22 findings.\nHistorical note:\nThis phase is withdrawn.\nCurrent status: Phase 2.5 is withdrawn.')).toEqual([]); -}); - - -test('a current Markdown status heading resets historical context for an owned withdrawal',()=>{ - const prefix='Phase 2.5 complete with 22 findings retained.\nHistorical note:\nThis phase is withdrawn.\n'; - expect(observe(prefix+'## Current status\nPhase 2.5 is withdrawn.')).toEqual([]); - expect(observe(prefix+'`## Current status`\nThis phase is withdrawn.')).toEqual([{phase:2.5,ts:at}]); -}); - -test('inline code around an owned status is scalar formatting while a whole quoted statement stays literal',()=>{ - const prefix='Phase 2.5 complete with 22 findings retained.\n'; - expect(observe(prefix+'This phase is `withdrawn`.')).toEqual([]); - for(const literal of ['`This phase is withdrawn.`','"This phase is withdrawn."','```text\nThis phase is withdrawn.\n```']) - expect(observe(prefix+literal)).toEqual([{phase:2.5,ts:at}]); -}); diff --git a/test/batching-permission-at.test.ts b/test/batching-permission-at.test.ts deleted file mode 100644 index 95cec5922..000000000 --- a/test/batching-permission-at.test.ts +++ /dev/null @@ -1,117 +0,0 @@ -import {test,expect} from 'bun:test'; -import fs from 'node:fs';import os from 'node:os';import path from 'node:path';import {pathToFileURL} from 'node:url'; -import {createFilePermissionRecorder,recordFilePermission,currentFilePermissionEpoch} from './helpers/plan-count-file-permission'; -import {createPlanCountPermissionGuard,classifyPlanCountFrame} from './helpers/claude-pty-runner'; -import {E2E_TOUCHFILES,selectTests}from'./helpers/touchfiles'; -import captured from './fixtures/batching-permission-at.json'; - -function renderPermissionScreen(expected: string, paths: Pick = path): string { - // The capture is already laid out at the runner's 120 columns. Replacing its - // path must reflow that menu line, otherwise the PTY hard-wraps words in half. - return captured.screen.split('\n').map(original => { - const line = original.replaceAll(path.posix.dirname(captured.expectedPath), paths.dirname(expected)) - .replaceAll(path.posix.basename(captured.expectedPath), paths.basename(expected)); - if (line === original || line.length <= 120) return line; - const indent = /^ */.exec(line)![0], lines: string[] = []; let current = indent; - for (const word of line.trim().split(/\s+/)) { - if (indent.length + word.length > 120) throw Error('Fixture path exceeds the permission panel width'); - if (current.length > indent.length && current.length + 1 + word.length > 120) { lines.push(current); current = indent; } - current += (current.length > indent.length ? ' ' : '') + word; - } - return [...lines, current].join('\n'); - }).join('\n'); -} - -function fixture(){ - const dir=fs.mkdtempSync(path.join(os.tmpdir(),'batch-permission-')),cwd=path.join(dir,'cwd'),config=path.join(dir,'.claude'),expected=path.join(dir,'report.md');fs.mkdirSync(cwd);fs.writeFileSync(expected,'original'); - const recorder=createFilePermissionRecorder(cwd,config,expected)!;const startedAt=Date.now()-1000; - const screen=renderPermissionScreen(expected); - const transcript:any={status:'ready',calls:[],assistantMessages:[{sessionId:'synthetic-epoch',text:'Reviewing',timestamp:new Date().toISOString()}]}; - const record=(name:string,id:string,extra={})=>recordFilePermission(JSON.stringify({hook_event_name:name,tool_name:'Edit',session_id:'synthetic-epoch',tool_use_id:id,cwd,transcript_path:path.join(config,'projects','owned','synthetic-epoch.jsonl'),tool_input:{file_path:expected},...extra}),recorder.file,cwd,config,expected); - const read=()=>currentFilePermissionEpoch(recorder.file,expected,cwd,config,startedAt,transcript,screen); - return{dir,cwd,config,expected,recorder,screen,transcript,record,read,close(){recorder.dispose();fs.rmSync(dir,{recursive:true,force:true})}}; -} - -test('retained retry has a valid permission panel and real previous completion without a pending ID',()=>{ - expect(classifyPlanCountFrame(captured.screen)).toBe('permission');expect(captured.priorCompletedEdit[0]!.name).toBe('Edit');expect(captured.priorCompletedEdit[1]!.isError).toBe(false); - expect(captured.pendingEditId).toBeNull();expect(captured.provenance.originalOutcome).toBe('timeout');expect(captured.provenance.paidOutcomeReclassified).toBe(false); - const guard=createPlanCountPermissionGuard();expect(guard(captured.screen,captured.lastMatchedDisplayCompletion)).toBe('grant');expect(guard(captured.screen,captured.lastMatchedDisplayCompletion)).toBe('handled'); -}); - -test('a substituted long fixture path reflows the menu without splitting permission words', () => { - const prefix = ' always allow access to ', suffix = ' for this '; - const directory = '/' + 'x'.repeat(120 - prefix.length - suffix.length - 3 - 1); - const rawLine = `${prefix}${directory}${suffix}session`; - expect(`${rawLine.slice(0, 120)}\n${rawLine.slice(120)}`).toContain('ses\nsion'); - for (const paths of [path.posix, path.win32]) { - const expected = paths.join(directory, 'report.md'); - const screen = renderPermissionScreen(expected, paths); - const menu = screen.slice(screen.indexOf(' Do you want to make this edit')); - expect(menu.split('\n').every(line => line.length <= 120)).toBe(true); - expect(menu).toContain(paths.dirname(expected)); - expect(menu).toContain('edit to report.md?'); - expect(menu).toMatch(/1\. Yes[\s\S]+2\. Yes,[\s\S]+3\. No/); - expect(createPlanCountPermissionGuard()(screen, captured.lastMatchedDisplayCompletion)).toBe('grant'); - } -}); - -test('synthetic hook epochs release only the later exact request after its predecessor succeeds',()=>{ - const f=fixture();try{const guard=createPlanCountPermissionGuard(),input=()=>guard(f.screen,captured.lastMatchedDisplayCompletion,f.read()); - expect(input()).toBe('handled');f.record('PreToolUse','first');expect(input()).toBe('grant');expect(input()).toBe('handled'); - f.record('PostToolUse','first');expect(input()).toBe('handled');f.record('PreToolUse','first');expect(input()).toBe('handled'); - f.record('PreToolUse','second');expect(input()).toBe('grant');expect(input()).toBe('handled');f.record('PostToolUse','first');expect(input()).toBe('handled'); - }finally{f.close()} -}); -for(const reason of ['failed','no-result','foreign-session','foreign-path','other-tool','sidechain'])test(`a later matching menu cannot replace ${reason} predecessor evidence`,()=>{ - const f=fixture();try{const guard=createPlanCountPermissionGuard(),input=()=>guard(f.screen,'',f.read());f.record('PreToolUse','first');expect(input()).toBe('grant'); - if(reason==='failed')f.record('PostToolUseFailure','first');else if(reason!=='no-result')f.record('PostToolUse','first',reason==='foreign-session'?{session_id:'foreign'}:reason==='foreign-path'?{tool_input:{file_path:path.join(f.dir,'foreign','report.md')}}:reason==='other-tool'?{tool_name:'Read'}:{agent_id:'child'}); - f.record('PreToolUse','second');expect(input()).toBe('handled'); - }finally{f.close()} -}); - -test('batching supplies permission scope without adding a report completion contract',()=>{ - const source=fs.readFileSync(path.join(import.meta.dir,'skill-e2e-plan-eng-multi-finding-batching.test.ts'),'utf8');expect(source).toContain('permissionPlanPath: planPath');expect(source).not.toContain('expectedPlanPath:'); - for(const file of ['test/batching-permission-at.test.ts','test/fixtures/batching-permission-at.json'])expect(selectTests([file],E2E_TOUCHFILES,[]).selected).toEqual(['plan-eng-multi-finding-batching']); -}); - -test.skipIf(process.platform==='win32')('real fake CLI observes two file epochs without imposing terminal report validation',async()=>{ - const dir=fs.mkdtempSync(path.join(os.tmpdir(),'batch-permission-pty-')),fake=path.join(dir,'fake-claude'),worker=path.join(dir,'worker.ts'),events=path.join(dir,'events.jsonl'),output=path.join(dir,'output.json'),expected=path.join(dir,'report.md');fs.writeFileSync(expected,'original'); - const screen=renderPermissionScreen(expected); - fs.writeFileSync(fake,`#!${process.execPath}\n`+String.raw` -import * as fs from 'node:fs';import * as path from 'node:path'; -const item=JSON.parse(process.env.FILE_EPOCH_CASE);const log=e=>fs.appendFileSync(item.events,JSON.stringify(e)+'\n'); -const sid='epoch-main';const nativePath=path.join(process.env.CLAUDE_CONFIG_DIR,'projects','epoch',sid+'.jsonl');fs.mkdirSync(path.dirname(nativePath),{recursive:true}); -const native=(role,content,extra={})=>fs.appendFileSync(nativePath,JSON.stringify({cwd:process.cwd(),sessionId:sid,isSidechain:false,timestamp:new Date().toISOString(),message:{role,content},...extra})+'\n'); -native('assistant',[{type:'text',text:'Reviewing fixture.'}]);log({type:'start',pid:process.pid,cwd:process.cwd()}); -const settings=JSON.parse(process.argv[process.argv.indexOf('--settings')+1]); -if(settings.hooks.PreToolUse[0].matcher!=='^ExitPlanMode$')throw Error('Exit recorder changed'); -const hook=async(name,id)=>{ - const entries=(settings.hooks[name]??[]).filter(h=>h.matcher==='^(Write|Edit)$'); - if(entries.length!==1)throw Error('Expected exactly one caller-owned file recorder'); - for(const entry of entries){ - const event={hook_event_name:name,tool_name:'Edit',session_id:sid,tool_use_id:id,cwd:process.cwd(),transcript_path:nativePath,tool_input:{file_path:item.activePlan?path.join(process.cwd(),'PLAN.md'):item.expected,old_string:'old',new_string:'new'}}; - const p=Bun.spawn(['bash','-c',entry.hooks[0].command],{stdin:new Blob([JSON.stringify(event)]),stdout:'pipe',stderr:'pipe'}); - const [code,out,err]=await Promise.all([p.exited,new Response(p.stdout).text(),new Response(p.stderr).text()]);if(code||out||err)throw Error('hook was not silent');log({type:'hook',name,id}); - } -}; -let stage='startup';const paint=()=>process.stdout.write('\x1b[2J\x1b[H'+item.screen.replaceAll('__ACTIVE_PLAN_PATH__',path.join(process.cwd(),'PLAN.md')).replaceAll('\n','\r\n')); -process.stdin.setRawMode?.(true);process.stdin.on('data',async data=>{ - const input=data.toString();log({type:'input',stage,input}); - if(stage==='startup'){stage='first';await hook('PreToolUse','first');paint();return;} - if(stage==='old-pane'||stage==='done'){log({type:'unexpected'});return;} - if(input!=='1\r')throw Error('default permission input changed'); - if(stage==='first'){stage='old-pane';await hook('PostToolUse','first');if(item.intervening){await hook('PreToolUse','automatic');await hook('PostToolUse','automatic');}paint();setTimeout(async()=>{await hook('PreToolUse','second');stage='second';paint();},3200);return;} - stage='done';await hook('PostToolUse','second'); - const q={header:'Finding',question:'Apply this repair?',options:[{label:'Fix'},{label:'Keep'}]}; - native('assistant',[{type:'tool_use',name:'AskUserQuestion',id:'finding',input:{questions:[q]}}]);native('user',[{type:'tool_result',tool_use_id:'finding',content:'Answered'}],{toolUseResult:{answers:{[q.question]:'Fix'}}}); - process.stdout.write('\x1b[2J\x1b[HCompletion summary\r\n'); -});process.on('SIGINT',()=>process.exit(0));process.stdin.resume();process.stdout.write('FILE_EPOCH_READY\r\n'); -`);fs.chmodSync(fake,0o755); - fs.writeFileSync(worker,`import {runPlanSkillCounting} from ${JSON.stringify(pathToFileURL(path.join(import.meta.dir,'helpers/claude-pty-runner.ts')).href)};const o=await runPlanSkillCounting({skillName:'plan-eng-review',slashCommand:'/plan-eng-review',followUpPrompt:'Review this disposable batching fixture.',permissionPlanPath:${JSON.stringify(expected)},startupReadyMarker:'FILE_EPOCH_READY',isLastStep0AUQ:()=>false,isReviewAUQ:()=>true,reviewCountCeiling:2,timeoutMs:28000,env:{FILE_EPOCH_CASE:${JSON.stringify(JSON.stringify({events,expected,screen}))}}});await Bun.write(${JSON.stringify(output)},JSON.stringify(o));`); - const child=Bun.spawn([process.execPath,worker],{env:{...process.env,BROWSE_TERMINAL_BINARY:fake,EVALS_HERMETIC:'1'},stdout:'pipe',stderr:'pipe'});const killer=setTimeout(()=>child.kill('SIGKILL'),33000); - try{const[code,out,err]=await Promise.all([child.exited,new Response(child.stdout).text(),new Response(child.stderr).text()]);expect(code,out+err).toBe(0); - const o=JSON.parse(fs.readFileSync(output,'utf8'));expect(o.outcome,JSON.stringify(o)).toBe('completion_summary');expect(o.reviewCount).toBe(1);expect(fs.readFileSync(expected,'utf8')).toBe('original'); - const rows=fs.readFileSync(events,'utf8').trim().split('\n').map(l=>JSON.parse(l));expect(rows.filter(e=>e.type==='input').map(e=>e.input)).toEqual(['/plan-eng-review\r','1\r','1\r']);expect(rows.some(e=>e.type==='unexpected')).toBe(false); - expect(()=>process.kill(rows[0].pid,0)).toThrow();expect(fs.existsSync(rows[0].cwd)).toBe(false); - }finally{clearTimeout(killer);child.kill('SIGKILL');if(fs.existsSync(events)){const first=JSON.parse(fs.readFileSync(events,'utf8').split('\n')[0]!);try{process.kill(first.pid,'SIGKILL');}catch{}}fs.rmSync(dir,{recursive:true,force:true});} -},35000); diff --git a/test/ceo-completion-handoff-m.test.ts b/test/ceo-completion-handoff-m.test.ts deleted file mode 100644 index 9b9254993..000000000 --- a/test/ceo-completion-handoff-m.test.ts +++ /dev/null @@ -1,43 +0,0 @@ -import { describe, expect, test } from 'bun:test'; -import * as fs from 'node:fs'; -import * as os from 'node:os'; -import * as path from 'node:path'; -import { hasNativePlanTerminal, - nativePlanCallFingerprint } from './helpers/claude-pty-runner'; -import type { NativePlanQuestionCall } from './helpers/plan-count-transcript'; -import captured from './fixtures/ceo-completion-handoff-m-call.json'; -import nextStepCapture from './fixtures/ceo-handoff-n-calls.json'; - -const calls = () => structuredClone(captured.calls) as NativePlanQuestionCall[]; -const handoff = () => calls().at(-1)!; -const fingerprint = (call: NativePlanQuestionCall) => nativePlanCallFingerprint(call, 0, false); -describe('CEO completion described by a native navigation choice', () => { -}); - -describe('native next-review navigation with a resolved CEO recap', () => { - const retryCalls = () => structuredClone(captured.distinctRetry.calls) as NativePlanQuestionCall[]; - const retryHandoff = () => retryCalls().at(-1)!; -}); - -describe('CEO completed next-step identity in native option order', () => { - const input = () => structuredClone(nextStepCapture.calls) as NativePlanQuestionCall[]; - const actual = () => input().at(-1)!; - test('the actual report precedes handoff but the captured absent Exit remains incomplete', () => { - const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'ceo-native-next-step-')); - const report = path.join(dir, 'plan.md'); - try { - fs.writeFileSync(report, nextStepCapture.report.content); - const written = Date.parse(nextStepCapture.report.successfulUpdateAt) / 1000; - fs.utimesSync(report, written, written); - const calls = input(); - expect(Date.parse(calls.at(-2)!.answeredAt!)).toBeLessThan(written * 1000); - expect(Date.parse(calls.at(-1)!.answeredAt!)).toBeGreaterThan(written * 1000); - const transcript = { status: 'ready' as const, calls, assistantMessages: [], - planReadyRequests: structuredClone(nextStepCapture.planReadyRequests) }; - const admin = new Set([fingerprint(calls.at(-1)!).signature]); - expect(hasNativePlanTerminal(transcript, report, Date.parse('2026-09-09T01:06:22Z'), 'plan_ready', admin)).toBe(false); - } finally { - fs.rmSync(dir, { recursive: true, force: true }); - } - }); -}); diff --git a/test/ceo-completion-handoff-o.test.ts b/test/ceo-completion-handoff-o.test.ts deleted file mode 100644 index d7361deff..000000000 --- a/test/ceo-completion-handoff-o.test.ts +++ /dev/null @@ -1,40 +0,0 @@ -import { describe, expect, test } from 'bun:test'; -import * as fs from 'node:fs'; -import * as os from 'node:os'; -import * as path from 'node:path'; -import { hasNativePlanTerminal } from './helpers/claude-pty-runner'; -import type { NativePlanQuestionCall } from './helpers/plan-count-transcript'; -import captured from './fixtures/ceo-completion-handoff-o-call.json'; -import capturedQ from './fixtures/ceo-completion-handoff-q-call.json'; - -const calls = () => structuredClone(captured.calls) as NativePlanQuestionCall[]; -const handoff = () => calls().at(-1)!; -describe('closed CEO navigation with the native review-prefixed identity', () => { -}); - -describe('CEO completion recap after native project metadata', () => { - const qCalls = () => structuredClone(capturedQ.calls) as NativePlanQuestionCall[]; - const qHandoff = () => qCalls().at(-1)!; - test('actual full report and Exit chronology retain last substantive-answer freshness', () => { - const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'ceo-metadata-navigation-')); - const report = path.join(dir, 'plan.md'); - try { - fs.writeFileSync(report, capturedQ.reportContent); - const reportAt = Date.parse(capturedQ.reportAt) / 1000; - fs.utimesSync(report, reportAt, reportAt); - const transcript = { status: 'ready' as const, calls: qCalls(), assistantMessages: [], planReadyRequests: structuredClone(capturedQ.planReadyRequests) }; - const administrative = new Set([`${qHandoff().sessionId}:${qHandoff().toolUseId}`]); - const start = Date.parse('2026-09-09T03:25:54Z'); - expect(hasNativePlanTerminal(transcript, report, start, 'plan_ready')).toBe(false); - expect(hasNativePlanTerminal(transcript, report, start, 'plan_ready', administrative)).toBe(true); - transcript.planReadyRequests[0]!.failed = true; - expect(hasNativePlanTerminal(transcript, report, start, 'plan_ready', administrative)).toBe(false); - transcript.planReadyRequests = structuredClone(capturedQ.planReadyRequests); - fs.utimesSync(report, start / 1000, start / 1000); - expect(hasNativePlanTerminal(transcript, report, start, 'plan_ready', administrative)).toBe(false); - fs.writeFileSync(report, 'Incomplete plan'); - fs.utimesSync(report, reportAt, reportAt); - expect(hasNativePlanTerminal(transcript, report, start, 'plan_ready', administrative)).toBe(false); - } finally { fs.rmSync(dir, { recursive: true, force: true }); } - }); -}); diff --git a/test/ceo-handoff-y.test.ts b/test/ceo-handoff-y.test.ts deleted file mode 100644 index cc2b77e3a..000000000 --- a/test/ceo-handoff-y.test.ts +++ /dev/null @@ -1,21 +0,0 @@ -import {describe,expect,test} from 'bun:test'; -import fs from 'node:fs'; -import os from 'node:os'; -import path from 'node:path'; -import fixture from './fixtures/ceo-handoff-y-call.json'; -import type {NativePlanQuestionCall} from './helpers/plan-count-transcript'; -import {hasNativePlanTerminal,nativePlanCallFingerprint} from './helpers/claude-pty-runner'; -const fp=(c:NativePlanQuestionCall)=>nativePlanCallFingerprint(c,0,false); -describe('Y bare next-Eng navigation is administrative, not completion evidence',()=>{ - test('independent fresh report and native Exit still gate completion; menu alone cannot pass',()=>{ - const dir=fs.mkdtempSync(path.join(os.tmpdir(),'gstack-handoff-y-free-'));const report=path.join(dir,'report.md'); - try{fs.writeFileSync(report,fixture.report);const calls=structuredClone(fixture.calls) as NativePlanQuestionCall[];const transcript={status:'ready' as const,calls,assistantMessages:[],planReadyRequests:structuredClone(fixture.planReadyRequests)};const handoff=calls.at(-1)!;const admin=new Set([fp(handoff).signature]);const issueAt=Date.parse(calls.at(-2)!.answeredAt!),handoffAt=Date.parse(handoff.answeredAt!);const started=Date.parse(calls[0]!.answeredAt!)-1000; - // Controlled metadata only: original Y report mtime was not captured. - const between=(issueAt+handoffAt)/2;fs.utimesSync(report,between/1000,between/1000); - expect(hasNativePlanTerminal(transcript,report,started,'plan_ready')).toBe(false);expect(hasNativePlanTerminal(transcript,report,started,'plan_ready',admin)).toBe(true); - fs.utimesSync(report,(issueAt-1)/1000,(issueAt-1)/1000);expect(hasNativePlanTerminal(transcript,report,started,'plan_ready',admin)).toBe(false); - fs.utimesSync(report,between/1000,between/1000);expect(hasNativePlanTerminal({...transcript,planReadyRequests:[]},report,started,'plan_ready',admin)).toBe(false); - expect(hasNativePlanTerminal({...transcript,calls:[handoff]},report,started,'plan_ready',admin)).toBe(false); - }finally{fs.rmSync(dir,{recursive:true,force:true});} - }); -}); diff --git a/test/ceo-hold-commitment-ar.test.ts b/test/ceo-hold-commitment-ar.test.ts deleted file mode 100644 index 1f3eab70a..000000000 --- a/test/ceo-hold-commitment-ar.test.ts +++ /dev/null @@ -1,105 +0,0 @@ -import { expect, test } from 'bun:test'; -import { hasNativePostAnswerCeoPosture, nativeCeoModeAnswer } from './helpers/ceo-mode-option'; -import type { PlanCountTranscript } from './helpers/plan-count-transcript'; -import { E2E_TOUCHFILES, selectTests } from './helpers/touchfiles'; -import captured from './fixtures/ceo-hold-commitment-ar.json'; - -const posture = /\b(rigor|bulletproof|hold\s*scope|maximum\s+rigor)\b/i; -const original = captured.transcript.assistantMessages[0]!.text; -const replay = () => structuredClone(captured.transcript) as PlanCountTranscript; -const matches = (transcript = replay()) => hasNativePostAnswerCeoPosture( - transcript, 'HOLD SCOPE', posture, captured.selectionStartedAt, -); -const withText = (text: string) => { const t = replay(); t.assistantMessages[0]!.text = text; return matches(t); }; - -test('the actual failed attempt adopted HOLD through scope, hardening and exclusion', () => { - expect(captured.provenance.actualState).toBe('failed'); - expect(nativeCeoModeAnswer(replay(), 'HOLD SCOPE', captured.selectionStartedAt)?.toolUseId) - .toBe('toolu_01E1HnYjRCz79826bo7nNnoK'); - expect(posture.test(original)).toBe(false); - expect(matches()).toBe(true); - // This is prospective posture recognition, not evidence of completed work. - for (const prefix of ["I'm keeping", 'I am keeping', 'I will keep', "We'll keep", 'We will keep', 'We are keeping']) { - expect(withText(original.replace("I'll keep", prefix)), prefix).toBe(true); - } - expect(withText(original.replace("I'll", 'I’ll').replace("PLAN.md's", 'PLAN.md’s'))).toBe(true); -}); - -test('explicitly future, conditional and quoted statements are not adopted current posture', () => { - for (const text of [ - original.replace("I'll keep", 'I will later keep'), - original.replace("I'll keep", 'I will eventually keep'), - original.replace("I'll keep", 'I would keep'), - original.replace("I'll keep", 'I may keep'), - original.replace("I'll keep", "I'll not keep"), - original.replace('scope fixed', 'scope tomorrow fixed'), - original.replace('production visibility', 'production visibility next week'), - original.replace('production visibility', 'production visibility tomorrow'), - ...['after approval', 'once approved', 'when approved', 'after launch', 'pending approval', 'subject to approval'].map(when => - original.replace('production visibility', 'production visibility ' + when)), - 'Later, ' + original, 'If you approve, ' + original, - 'Hypothetical scenario. ' + original, 'Example only: ' + original, - '"' + original + '"', '> ' + original, - '```text\n' + original + '\n```', '~~~text\n' + original + '\n~~~', - 'Read(file)\n' + original, 'The user said: ' + original, - ]) expect(withText(text), text).toBe(false); -}); - -test('all three obligations remain concrete and bound to the selected plan', () => { - for (const [from, to] of [ - ['PLAN.md', 'OTHER.md'], ['PLAN.md', 'archive/PLAN.md'], - ["PLAN.md's four bullets plus the approved schema", 'the future expanded plan'], - ['plus the approved schema', 'plus a new unapproved schema'], - [', pressure-testing every stated behavior for failure modes, errors, tests, and production visibility', ''], - ['errors, tests, and production visibility', 'word choice and formatting'], - ['while deferring anything extra rather than adding it silently', 'while adding anything extra'], - ['while deferring', 'while not deferring'], ['pressure-testing', 'not pressure-testing'], - ]) expect(withText(original.replace(from!, to!)), from).toBe(false); - for (const contextChange of [ - (text: string) => text.replace('PLAN.md', 'PLAN.md and OTHER.md'), - (text: string) => text.replace('schema) approved', 'schema) not approved'), - (text: string) => text.replace('schema) approved', 'schema) discussed'), - ...['approved if the user agrees', 'approved once migration finishes', 'approved pending migration', 'approved subject to migration'].map(status => - (text: string) => text.replace('schema) approved', 'schema) ' + status)), - ]) { - const t = replay(); const q = t.calls[0]!.questions[0]!; const prior = q.question; - q.question = contextChange(q.question); t.calls[0]!.answers = { [q.question]: t.calls[0]!.answers![prior]! }; - expect(matches(t)).toBe(false); - } -}); - -test('current corrections withdraw a commitment; quoted corrections do not', () => { - for (const correction of [ - 'Correction: I will expand scope to include defaults.', - 'Correction: I will not keep scope fixed to these requirements.', - 'Correction: I am no longer keeping scope to those requirements.', - 'The formerly excluded additions are in scope.', - ]) { - expect(withText(original + '\n\n' + correction), correction).toBe(false); - for (const quote of ['> ' + correction, '```text\n' + correction + '\n```', '~~~text\n' + correction + '\n~~~', 'A quotation: "' + correction + '"']) { - expect(withText(original + '\n\n' + quote), quote).toBe(true); - } - } -}); - -test('native selection, session and timestamp evidence remain required', () => { - for (const change of [ - (t: PlanCountTranscript) => { t.status = 'missing'; }, - (t: PlanCountTranscript) => { t.calls[0]!.answered = false; }, - (t: PlanCountTranscript) => { t.calls[0]!.failed = true; }, - (t: PlanCountTranscript) => { t.calls[0]!.answeredAt = new Date(captured.selectionStartedAt - 1).toISOString(); }, - (t: PlanCountTranscript) => { t.calls[0]!.answers![t.calls[0]!.questions[0]!.question] = 'Scope expansion'; }, - (t: PlanCountTranscript) => { t.calls[0]!.answers![t.calls[0]!.questions[0]!.question] = 'Unknown'; }, - (t: PlanCountTranscript) => { t.assistantMessages[0]!.sessionId = 'foreign'; }, - (t: PlanCountTranscript) => { t.assistantMessages[0]!.timestamp = t.calls[0]!.answeredAt!; }, - (t: PlanCountTranscript) => { t.assistantMessages[0]!.timestamp = 'invalid'; }, - (t: PlanCountTranscript) => { t.assistantMessages[0]!.timestamp = new Date(Date.now() + 60_000).toISOString(); }, - (t: PlanCountTranscript) => { t.assistantMessages = []; }, - ]) { const t = replay(); change(t); expect(matches(t)).toBe(false); } -}); - -test('new posture evidence selects the existing mode owner', () => { - for (const file of ['test/ceo-hold-commitment-ar.test.ts', 'test/fixtures/ceo-hold-commitment-ar.json']) { - expect(selectTests([file], E2E_TOUCHFILES, []).selected).toEqual(['plan-ceo-mode-routing']); - } -}); diff --git a/test/ceo-hold-posture-ag.test.ts b/test/ceo-hold-posture-ag.test.ts deleted file mode 100644 index 397768d2b..000000000 --- a/test/ceo-hold-posture-ag.test.ts +++ /dev/null @@ -1,249 +0,0 @@ -import { expect, test } from 'bun:test'; -import { hasNativePostAnswerCeoPosture, nativeCeoModeAnswer } from './helpers/ceo-mode-option'; -import type { PlanCountTranscript } from './helpers/plan-count-transcript'; -import captured from './fixtures/ceo-hold-posture-ag.json'; -import { E2E_TOUCHFILES, selectTests } from './helpers/touchfiles'; - -const posture = /\b(rigor|bulletproof|hold\s*scope|maximum\s+rigor)\b/i; -const original = captured.transcript.assistantMessages[0]!.text; -const replay = () => structuredClone(captured.transcript) as PlanCountTranscript; -const matches = (transcript = replay()) => hasNativePostAnswerCeoPosture( - transcript, 'HOLD SCOPE', posture, captured.selectionStartedAt, -); - -test('the captured selected HOLD scope lock and hardening establish posture without a keyword', () => { - const transcript = replay(); - expect(captured.provenance.actualState).toBe('failed'); - expect(nativeCeoModeAnswer(transcript, 'HOLD SCOPE', captured.selectionStartedAt)?.toolUseId) - .toBe('toolu_011bt3yabPDSEsPNm97EhqV4'); - expect(posture.test(original)).toBe(false); - expect(matches(transcript)).toBe(true); -}); - -test('ordinary current scope declarations preserve the same three obligations', () => { - for (const text of [ - original.replace("I'm locking", 'I will lock'), - original.replace("I'm locking", "I'll lock"), - original.replace("I'm locking", 'We are keeping').replace('the four PLAN.md bullets from approach B', 'the agreed plan') - .replace('flagging anything beyond', 'treating everything outside').replace('hunting', 'checking'), - original.replace("I'm locking", 'I am holding').replace('four PLAN.md bullets from approach B', 'PLAN.md requirements') - .replace('flagging', 'marking').replace('hunting', 'looking'), - original.replace("I'm", 'I’m').replace('PLAN.md', '**PLAN.md**'), - ]) { - const transcript = replay(); transcript.assistantMessages[0]!.text = text; - expect(matches(transcript)).toBe(true); - } -}); - -test('deferred commitments, conditions and quotation cannot establish the current posture', () => { - for (const text of [ - original.replace("I'm locking", 'I would lock'), - original.replace("I'm locking", 'I will later lock'), - 'If you approve, ' + original, - 'Later, ' + original, - 'Example only: ' + original, - 'An unproven hypothesis: ' + original, - 'Example only. ' + original, - '"' + original + '"', - '> ' + original, - '```text\n' + original + '\n```', - '~~~~\n' + original + '\n~~~~', - 'Read(file)\n' + original, - 'The user said: ' + original, - original.replace('and hunting', 'and not hunting'), - ]) { - const transcript = replay(); transcript.assistantMessages[0]!.text = text; - expect(matches(transcript), text).toBe(false); - } -}); - -test('all three obligations refer to the selected current scope', () => { - for (const text of [ - original.replace('PLAN.md', 'OTHER.md'), - original.replace('PLAN.md', 'archive/PLAN.md'), - original.replace('the four PLAN.md bullets from approach B', 'the future expanded plan'), - original.replace('the four PLAN.md bullets from approach B', 'the two imagined requirements'), - original.replace('out of scope', 'in scope'), - original.replace('as out of scope', 'as not out of scope'), - original.replace('flagging anything beyond that (defaults, sharing, deep links) as out of scope, and ', ''), - original.replace(/, and hunting[^.]+\./, '.'), - original.replace('constraints, error handling, UI edge cases, access-rule leaks', 'word choice and formatting'), - original + ' I am expanding scope to include a new feature.', - original + ' I am adding extra features to scope.', - ]) { - const transcript = replay(); transcript.assistantMessages[0]!.text = text; - expect(matches(transcript), text).toBe(false); - } - const ambiguous = replay(); - const question = ambiguous.calls[0]!.questions[0]!; - const oldQuestion = question.question; - question.question = question.question.replace('reviewing PLAN.md', 'reviewing PLAN.md and OTHER.md'); - ambiguous.calls[0]!.answers = { [question.question]: ambiguous.calls[0]!.answers![oldQuestion]! }; - expect(matches(ambiguous)).toBe(false); -}); - -test('explicit later corrections withdraw scope locking, while quoted examples do not', () => { - const corrections = [ - 'Correction: the previously excluded defaults, sharing, and deep links are now in scope.', - 'Correction: I am no longer locking scope to those requirements.', - 'I am not keeping scope to those requirements.', - 'The formerly excluded additions are in scope.', - ]; - for (const correction of corrections) { - const transcript = replay(); - transcript.assistantMessages[0]!.text = original + '\n\n' + correction; - expect(matches(transcript), correction).toBe(false); - for (const quote of ['> ' + correction, '```text\n' + correction + '\n```', - '~~~text\n' + correction + '\n~~~', 'An example of withdrawn wording is: "' + correction + '"']) { - transcript.assistantMessages[0]!.text = original + '\n\n' + quote; - expect(matches(transcript), quote).toBe(true); - } - } -}); - -test('only a real selected HOLD answer followed by its own public statement supplies evidence', () => { - for (const change of [ - (t: PlanCountTranscript) => { t.status = 'missing'; }, - (t: PlanCountTranscript) => { t.calls[0]!.answered = false; }, - (t: PlanCountTranscript) => { t.calls[0]!.failed = true; }, - (t: PlanCountTranscript) => { t.calls[0]!.answeredAt = new Date(captured.selectionStartedAt - 1).toISOString(); }, - (t: PlanCountTranscript) => { t.calls[0]!.answers![t.calls[0]!.questions[0]!.question] = 'Scope Expansion'; }, - (t: PlanCountTranscript) => { t.calls[0]!.answers![t.calls[0]!.questions[0]!.question] = 'Unknown'; }, - (t: PlanCountTranscript) => { t.assistantMessages[0]!.sessionId = 'foreign'; }, - (t: PlanCountTranscript) => { t.assistantMessages[0]!.timestamp = t.calls[0]!.answeredAt!; }, - (t: PlanCountTranscript) => { t.assistantMessages[0]!.timestamp = 'invalid'; }, - (t: PlanCountTranscript) => { t.assistantMessages[0]!.timestamp = new Date(Date.now() + 60_000).toISOString(); }, - (t: PlanCountTranscript) => { t.assistantMessages = []; }, - ]) { - const transcript = replay(); change(transcript); expect(matches(transcript)).toBe(false); - } - const expansion = replay(); - expansion.calls[0]!.answers![expansion.calls[0]!.questions[0]!.question] = 'Scope Expansion'; - expect(hasNativePostAnswerCeoPosture(expansion, 'SCOPE EXPANSION', posture, captured.selectionStartedAt)).toBe(false); -}); - -test('new evidence controls select only the existing mode paid owner', () => { - for (const file of ['test/ceo-hold-posture-ag.test.ts', 'test/fixtures/ceo-hold-posture-ag.json']) { - expect(Object.entries(E2E_TOUCHFILES).filter(([, files]) => files.includes(file)).map(([owner]) => owner)) - .toEqual(['plan-ceo-mode-routing']); - expect(selectTests([file], E2E_TOUCHFILES, []).selected).toEqual(['plan-ceo-mode-routing']); - } -}); - -// Exact public AY parent narration after the answered HOLD SCOPE mode AUQ. -// Its native ownership controls use the existing PLAN.md / approved-approach-B fixture. -const ambiguityNarration = "I'm holding strictly to the plan's approved scope (Approach B, private-only views) and flagging any ambiguities the sketch leaves undecided as targeted questions rather than expanding scope. First up: what happens when a saved view's filters reference something that's been deleted.\n\n"; -const ambiguityReplay = () => { - const transcript = replay(); - transcript.assistantMessages[0]!.text = ambiguityNarration; - return transcript; -}; -const ambiguityMatches = (text = ambiguityNarration) => { - const transcript = ambiguityReplay(); transcript.assistantMessages[0]!.text = text; - return matches(transcript); -}; - -test('approved scope plus targeted ambiguity questions applies HOLD without naming the mode', () => { - expect(posture.test(ambiguityNarration)).toBe(false); - expect(ambiguityMatches()).toBe(true); - for (const text of [ - ambiguityNarration.replace("I'm holding", 'We are keeping'), - ambiguityNarration.replace("I'm holding", 'I will hold'), - ambiguityNarration.replace('the sketch leaves undecided', 'in the plan').replace('flagging', 'surfacing'), - ambiguityNarration.replace("plan's", "PLAN.md's"), - ambiguityNarration.replace("I'm", 'I’m').replace("plan's", 'plan’s'), - ]) expect(ambiguityMatches(text), text).toBe(true); -}); - -test('ambiguity wording must adopt every obligation without quoting, negating or deferring it', () => { - for (const text of [ - '> ' + ambiguityNarration, '"' + ambiguityNarration.trim() + '"', - '```text\n' + ambiguityNarration + '```', '~~~text\n' + ambiguityNarration + '~~~', - 'Example only: ' + ambiguityNarration, 'The user said: ' + ambiguityNarration, - 'Read(file)\n' + ambiguityNarration, 'If approved, ' + ambiguityNarration, - ambiguityNarration.replace("I'm holding", 'I would hold'), - ambiguityNarration.replace("I'm holding", 'I will later hold'), - ambiguityNarration.replace("I'm holding", "I'm not holding"), - ambiguityNarration.replace('and flagging', 'and not flagging'), - ambiguityNarration.replace('approved scope', 'proposed scope'), - ambiguityNarration.replace("plan's", "OTHER.md's"), - ambiguityNarration.replace('private-only views', 'OTHER.md views'), - ambiguityNarration.replace('Approach B', 'Approach C'), - ambiguityNarration.replace('as targeted questions rather than expanding scope', 'as optional improvements'), - ambiguityNarration.replace('rather than expanding scope', 'while expanding scope'), - ambiguityNarration.replace('ambiguities the sketch leaves undecided', 'word choice and formatting'), - ]) expect(ambiguityMatches(text), text).toBe(false); - for (const correction of [ - 'I am expanding scope to include sharing.', - 'Correction: I will add defaults to scope.', - 'The previously excluded sharing feature is now in scope.', - 'Correction: I am no longer holding scope to this plan.', - "Correction: I am not holding strictly to the plan's approved scope.", - 'Correction: I am no longer flagging ambiguities as targeted questions.', - 'Correction: this posture is withdrawn.', - 'This posture is no longer current.', - ]) { - expect(ambiguityMatches(ambiguityNarration + correction), correction).toBe(false); - expect(ambiguityMatches(ambiguityNarration + '> ' + correction), correction).toBe(true); - } -}); - -test('ambiguity posture stays bound to the approved plan and its actual native answer', () => { - for (const change of [ - (t: PlanCountTranscript) => { t.calls[0]!.answered = false; }, - (t: PlanCountTranscript) => { t.calls[0]!.failed = true; }, - (t: PlanCountTranscript) => { t.calls[0]!.answers![t.calls[0]!.questions[0]!.question] = 'Scope Expansion'; }, - (t: PlanCountTranscript) => { t.assistantMessages[0]!.sessionId = 'foreign'; }, - (t: PlanCountTranscript) => { t.assistantMessages[0]!.timestamp = t.calls[0]!.answeredAt!; }, - (t: PlanCountTranscript) => { t.assistantMessages[0]!.timestamp = 'invalid'; }, - (t: PlanCountTranscript) => { t.assistantMessages[0]!.timestamp = new Date(Date.now() + 60_000).toISOString(); }, - ]) { const transcript = ambiguityReplay(); change(transcript); expect(matches(transcript)).toBe(false); } - for (const [from, to] of [ - ['PLAN.md', 'PLAN.md and OTHER.md'], - ['approved.', 'not approved.'], - ['approved.', 'approved if accepted.'], - ['approved.', 'discussed.'], - ]) { - const transcript = ambiguityReplay(); const q = transcript.calls[0]!.questions[0]!; - const before = q.question; q.question = before.replace(from!, to!); - transcript.calls[0]!.answers = { [q.question]: transcript.calls[0]!.answers![before]! }; - expect(matches(transcript), to).toBe(false); - } -}); - -import retainedPreservationCaptures from './fixtures/ceo-hold-preservation-f359.json'; -{ -const captures = retainedPreservationCaptures; -const posture=/\b(rigor|bulletproof|hold\s*scope|maximum\s+rigor)\b/i; -const clone=(i=0)=>structuredClone(captures[i]) as any; -const check=(x:any)=>hasNativePostAnswerCeoPosture(x.transcript,'HOLD SCOPE',posture,x.selectionStartedAt,x.tools,x.source); -const decision=(x:any)=>x.transcript.calls.find((c:any)=>c.questions[0]?.question.match(/^D\d+ — Keep/)); -function editQuestion(x:any,change:(q:any)=>void){const c=decision(x);const before=c.questions[0].question;change(c.questions[0]);const after=c.questions[0].question;if(before!==after){c.answers[after]=c.answers[before];delete c.answers[before]};x.tools.find((t:any)=>t.kind==='use'&&t.toolUseId===c.toolUseId).input.questions=structuredClone(c.questions)} -for(let i=0;i<2;i++)test(`actual acknowledged preserve decision ${i+1}`,()=>{const x=clone(i);expect(check(x)).toBe(true)}); -const mutations:Recordvoid>={ - 'unanswered':x=>{decision(x).answered=false}, - 'failed answer':x=>{x.tools.find((t:any)=>t.kind==='result'&&t.toolUseId===decision(x).toolUseId).isError=true}, - 'unmatched native request':x=>{x.tools.find((t:any)=>t.kind==='use'&&t.toolUseId===decision(x).toolUseId).input.questions=[]}, - 'foreign decision session':x=>{decision(x).sessionId='foreign'}, - 'foreign source path':x=>{x.source.path='/foreign/PLAN.md'}, - 'altered source bytes':x=>{x.source.content=x.source.content.replace('update,','share,')}, - 'different named source':x=>{editQuestion(x,q=>q.question=q.question.replace('PLAN.md','OTHER.md'))}, - 'unrelated choice':x=>{editQuestion(x,q=>{q.question=q.question.replaceAll('update','sharing');q.options=q.options.map((o:any)=>({...o,label:o.label.replaceAll('update','sharing')}))});const c=decision(x);c.answers[c.questions[0].question]=c.questions[0].options[0].label}, - 'expanding description':x=>{editQuestion(x,q=>q.options[0].description+=' Also add shared team views outside the plan.')}, - 'mere mode label':x=>{editQuestion(x,q=>{q.question=q.question.replace(/ELI10:[\s\S]*?Stakes if/,'ELI10: Keep it.\nStakes if').replace(/Stakes if[\s\S]*?Recommendation:/,'Stakes if we pick wrong: None.\nRecommendation:');q.options.forEach((o:any)=>o.description='Fine.')})}, - 'historical decision':x=>{editQuestion(x,q=>q.question='Historical example: '+q.question)}, - 'quoted decision':x=>{editQuestion(x,q=>q.question=q.question.split('\n').map((l:string)=>'> '+l).join('\n'))}, - 'withdrawn decision':x=>{editQuestion(x,q=>q.question=q.question.replace('HOLD SCOPE review','withdrawn HOLD SCOPE review'))}, - 'later withdrawal':x=>{x.transcript.assistantMessages.push({sessionId:decision(x).sessionId,timestamp:new Date().toISOString(),text:'I withdraw this decision.'})}, - 'later scope expansion':x=>{x.transcript.assistantMessages.push({sessionId:decision(x).sessionId,timestamp:new Date().toISOString(),text:'I expand the scope.'})}, - 'missing source ACK':x=>{x.tools=x.tools.filter((t:any)=>!(t.kind==='result'&&x.tools.some((u:any)=>u.kind==='use'&&u.toolUseId===t.toolUseId&&u.name==='Read'&&u.input?.file_path===x.source.path)))}, - 'wrong actual choice':x=>{const c=decision(x);c.answers[c.questions[0].question]=c.questions[0].options[1].label}, -}; -for(const [name,mutate] of Object.entries(mutations))test(name,()=>{const x=clone();mutate(x);expect(check(x)).toBe(false)}); -test('later quoted withdrawal is not current withdrawal',()=>{const x=clone();x.transcript.assistantMessages.push({sessionId:decision(x).sessionId,timestamp:new Date().toISOString(),text:'Example: "I withdraw this decision."'});expect(check(x)).toBe(true)}); -test('new proof path is unavailable without explicit fixture source binding',()=>{const x=clone();expect(hasNativePostAnswerCeoPosture(x.transcript,'HOLD SCOPE',posture,x.selectionStartedAt,x.tools)).toBe(false)}); - -test('retry source cat requires the actual owned project',()=>{const x=clone(1);x.tools.find((t:any)=>t.kind==='use'&&t.input?.command?.includes('cat PLAN.md')).input.command=x.tools.find((t:any)=>t.kind==='use'&&t.input?.command?.includes('cat PLAN.md')).input.command.replace(x.source.path.replace('/PLAN.md',''),'/foreign');expect(check(x)).toBe(false)}); -test('retry source read ACK cannot be missing',()=>{const x=clone(1);const use=x.tools.find((t:any)=>t.kind==='use'&&t.input?.command?.includes('cat PLAN.md'));x.tools=x.tools.filter((t:any)=>!(t.kind==='result'&&t.toolUseId===use.toolUseId));expect(check(x)).toBe(false)}); - -} diff --git a/test/ceo-mode-colon-at.test.ts b/test/ceo-mode-colon-at.test.ts deleted file mode 100644 index 6d4b769c1..000000000 --- a/test/ceo-mode-colon-at.test.ts +++ /dev/null @@ -1,108 +0,0 @@ -import { describe, expect, test } from 'bun:test'; -import { findCeoModeOption, nativeCeoModeAnswer, nextCeoModeNavigation } from './helpers/ceo-mode-option'; -import type { PlanCountTranscript } from './helpers/plan-count-transcript'; -import { E2E_TOUCHFILES } from './helpers/touchfiles-data'; -import { selectTests } from './helpers/touchfiles'; -import captured from './fixtures/ceo-mode-colon-at.json'; - -function transcript(): PlanCountTranscript { - return { status: 'ready', calls: [structuredClone(captured)], assistantMessages: [] }; -} - -describe('CEO colon-prefixed native mode choices', () => { - test('the exact public menu resolves each named mode by display position', () => { - const options = captured.questions[0]!.options.map((option, i) => ({ index: i + 1, label: option.label })); - expect(findCeoModeOption(options, 'SELECTIVE EXPANSION')).toBe(1); - expect(findCeoModeOption(options, 'SCOPE EXPANSION')).toBe(2); - expect(findCeoModeOption(options, 'HOLD SCOPE')).toBe(3); - expect(findCeoModeOption(options, 'SCOPE REDUCTION')).toBe(4); - }); - - test('navigation selects expansion in either display order without changing native input', () => { - for (const reverse of [false, true]) { - const call = transcript().calls[0]!; - call.answered = false; - delete call.answers; - delete call.unansweredQuestionIndices; - const question = call.questions[0]!; - if (reverse) question.options.reverse(); - const original = structuredClone(call); - const visible = `☐ ${question.header}\n${question.question}\n` + question.options.map((option, i) => - `${i ? ' ' : '❯'} ${i + 1}. ${option.label}`).join('\n') + - '\nEnter to select · ↑/↓ to navigate · Esc to cancel'; - const action = nextCeoModeNavigation(visible, 'SCOPE EXPANSION', new Set(), call); - expect(action.kind).toBe('mode'); - expect(action.kind === 'mode' && action.index).toBe(reverse ? 3 : 2); - expect(call).toEqual(original); - } - }); - - test('the recorded wrong selection remains selective expansion, never expansion coverage', () => { - const actual = transcript(); - expect(nativeCeoModeAnswer(actual, 'SELECTIVE EXPANSION', 0)?.toolUseId) - .toBe('toolu_01XY3qPeSuJZa3H2uCfatJ8b'); - expect(nativeCeoModeAnswer(actual, 'SCOPE EXPANSION', 0)).toBeNull(); - expect(actual.calls[0]).toEqual(captured); - }); - - test('pending, failed, stale and ambiguous native answers cannot prove selection', () => { - for (const change of [ - (value: PlanCountTranscript) => { value.calls[0]!.answered = false; }, - (value: PlanCountTranscript) => { value.calls[0]!.failed = true; }, - (value: PlanCountTranscript) => { delete value.calls[0]!.answers; }, - (value: PlanCountTranscript) => { value.calls[0]!.answeredAt = 'invalid'; }, - (value: PlanCountTranscript) => { - value.calls[0]!.questions[0]!.options.push({ label: 'E: SELECTIVE EXPANSION' }); - }, - ]) { - const value = transcript(); - change(value); - expect(nativeCeoModeAnswer(value, 'SELECTIVE EXPANSION', 0)).toBeNull(); - } - expect(nativeCeoModeAnswer(transcript(), 'SELECTIVE EXPANSION', Date.parse(captured.answeredAt) + 1)).toBeNull(); - const laterAmbiguous = transcript(); - const later = structuredClone(laterAmbiguous.calls[0]!); - later.toolUseId = 'later-ambiguous-mode'; - later.answeredAt = new Date(Date.parse(captured.answeredAt) + 1000).toISOString(); - later.questions[0]!.options.push({ label: 'E: SELECTIVE EXPANSION' }); - laterAmbiguous.calls.push(later); - expect(nativeCeoModeAnswer(laterAmbiguous, 'SELECTIVE EXPANSION', 0)).toBeNull(); - }); - - test('action titles, lookalikes and preview descriptions do not become modes', () => { - for (const label of [ - 'A: Use HOLD SCOPE for the next review', - 'B: Explain SCOPE EXPANSION', - 'AA: HOLD SCOPE', - '1: HOLD SCOPE', - 'A:: HOLD SCOPE', - 'A: HOLD SCOPES', - 'A: Fix contrast │ HOLD SCOPE', - 'A: Fix contrast ┌ SCOPE EXPANSION', - 'A: "HOLD SCOPE"', - 'Prior: HOLD SCOPE', - ]) expect(findCeoModeOption([{ index: 1, label }], 'HOLD SCOPE')).toBeNull(); - expect(() => findCeoModeOption([ - { index: 1, label: 'A: SELECTIVE EXPANSION │ SCOPE EXPANSION' }, - { index: 2, label: 'B: HOLD SCOPE' }, - ], 'SCOPE EXPANSION')).toThrow('not in option labels'); - }); - - test('duplicate and missing mode titles fail before selection; legacy prefixes still work', () => { - expect(() => findCeoModeOption([ - { index: 1, label: 'A: HOLD SCOPE' }, - { index: 2, label: 'HOLD SCOPE (recommended)' }, - ], 'HOLD SCOPE')).toThrow('duplicate'); - expect(() => findCeoModeOption([{ index: 1, label: 'A: SCOPE REDUCTION' }], 'HOLD SCOPE')) - .toThrow('not in option labels'); - for (const label of ['A) HOLD SCOPE', 'A. HOLD SCOPE', 'a: hold scope', 'A: HOLD SCOPE']) { - expect(findCeoModeOption([{ index: 3, label }], 'HOLD SCOPE')).toBe(3); - } - }); - - test('the exact public fixture and regression select the mode-routing workflow', () => { - for (const file of ['test/ceo-mode-colon-at.test.ts', 'test/fixtures/ceo-mode-colon-at.json']) { - expect(selectTests([file], E2E_TOUCHFILES, []).selected).toEqual(['plan-ceo-mode-routing']); - } - }); -}); diff --git a/test/ceo-mode-full-ad.test.ts b/test/ceo-mode-full-ad.test.ts deleted file mode 100644 index f680ea3a0..000000000 --- a/test/ceo-mode-full-ad.test.ts +++ /dev/null @@ -1,726 +0,0 @@ -import {describe,expect,test} from 'bun:test'; -import fs from 'node:fs';import os from 'node:os';import path from 'node:path'; -import {ceoExpansionPacingChoice,ceoExpansionPacingReady,ceoModeSubmissionInput,hasNativePostAnswerCeoPosture,nextCeoModeNavigation,nextCeoPostureContinuation} from './helpers/ceo-mode-option'; -import {capturePlanCountQuestion,nativePlanCallFingerprint,planCountPrerequisitePick,planCountQuestionInput,isNumberedOptionListVisible,isPlanReadyVisible} from './helpers/claude-pty-runner'; -import {readPlanCountTranscript,type NativePublicToolEvent,type NativePlanQuestionCall} from './helpers/plan-count-transcript'; -import captured from './fixtures/ceo-mode-full-ad.json'; -import kindCapture from './fixtures/ceo-expansion-posture-kind-dacc.json'; -import pauseCapture from './fixtures/ceo-expansion-pause-6714.json'; -import {E2E_TOUCHFILES,selectTests} from './helpers/touchfiles'; -const pattern=/\b(expansion|10x|delight|dream|cathedral|opt[\s-]?in)\b/i; -function replay(i:number){ - const item=captured.cases[i]!,root=fs.mkdtempSync(path.join(os.tmpdir(),'ceo-full-ad-')); - fs.mkdirSync(path.join(root,'projects','owned'),{recursive:true}); - fs.writeFileSync(path.join(root,'projects','owned',item.process.sessionId+'.jsonl'),item.records.map(r=>JSON.stringify(r)).join('\n')+'\n'); - const events:NativePublicToolEvent[]=[]; - try{return {item,transcript:readPlanCountTranscript(root,item.process.cwd,e=>events.push(e)),events};} - finally{fs.rmSync(root,{recursive:true,force:true});} -} -function pending(){const c=structuredClone(replay(0).transcript.calls[0]!);c.answered=false;delete c.answers;delete c.answeredAt;delete c.unansweredQuestionIndices;return c;} -// Full panes projected from exact native questions, not retained historical viewports. -function pane(call:NativePlanQuestionCall,index:number){const q=call.questions[index]!;return [ - call.questions.length>1?'← '+call.questions.map((v,i)=>`${i`${i?' ':'❯'} ${i+1}. ${v.label}`), - `Enter to select · ${call.questions.length>1?'Tab/Arrow keys':'↑/↓'} to navigate · Esc to cancel`].join('\n');} -function frame(c:NativePlanQuestionCall,index:number){const visible=pane(c,index);return {visible,active:capturePlanCountQuestion(visible,new Set(),0,true,c)!,routing:nativePlanCallFingerprint(c,0,true)};} -function match(e= replay(1)){return hasNativePostAnswerCeoPosture(e.transcript,'SCOPE EXPANSION',pattern,e.item.selectedAt!,e.events);} -function rebind(e:ReturnType){const d=e.transcript.calls[1]!,q=d.questions[0]!;e.events[2]!.input={questions:d.questions};d.answers={[q.question]:q.options[0]!.label};} -describe('full AD mode failures retain their actual outcomes',()=>{ - test('Proposal 1 is a completed scope decision after the actual selected mode',()=>{ - const e=replay(1);expect(e.item.actualState).toBe('failed');expect(e.transcript.calls).toHaveLength(2);expect(e.events).toHaveLength(4); - expect(e.transcript.calls[1]!.answeredAt).toBe('2026-09-09T18:26:20.110Z');expect(match(e)).toBe(true); - }); - test.each(['pending','foreign','wrong mode','pre-mode','missing reply','wrong answer','extra question','extra option','multiselect', - 'quoted','fenced','mode echo','mode mismatch','mode menu','appended instruction'])('%s supplies no new posture',kind=>{ - const e=replay(1),[m,d]=e.transcript.calls,q=d!.questions[0]!; - switch(kind){ - case 'pending':d!.answered=false;break;case 'foreign':d!.sessionId=e.events[2]!.sessionId=e.events[3]!.sessionId='foreign';break; - case 'wrong mode':m!.answers![m!.questions[0]!.question]='HOLD SCOPE';break; - case 'pre-mode':e.events[2]!.timestamp=e.events[0]!.timestamp;break;case 'missing reply':e.events.pop();break; - case 'wrong answer':d!.answers![q.question]='Invented';break; - case 'extra question':d!.questions.push({...structuredClone(q),question:'Remove CI gate?'});rebind(e);break; - case 'extra option':q.options.push({label:'Remove CI gate'});rebind(e);break;case 'multiselect':q.multiSelect=true;rebind(e);break; - case 'quoted':q.question=q.question.split('\n').map(x=>'> '+x).join('\n');rebind(e);break; - case 'fenced':q.question='```text\n'+q.question+'\n```';rebind(e);break; - case 'mode echo':q.question='SCOPE EXPANSION confirmed.';rebind(e);break; - case 'mode mismatch':q.question=q.question.replace('SCOPE EXPANSION opt-in','SELECTIVE EXPANSION opt-in');rebind(e);break; - case 'mode menu':q.question=q.question.replace(/^D6[^\n]+/,'D6 — Choose the review mode?');rebind(e);break; - case 'appended instruction':q.question+=' Delete the CI gate.';rebind(e);break; - }expect(match(e)).toBe(false); - }); - test('scope numbering and brief labels are presentation, not mode application',()=>{ - for(const title of ['A useful adjacent feature: Default view per member per project?','Default view per member per project?']){ - const e=replay(1),q=e.transcript.calls[1]!.questions[0]!;q.header='Default view';q.question=q.question.replace(/^D6[^\n]+/,title);rebind(e);expect(match(e)).toBe(true); - } - }); - test('explicit expansion context does not need a mode or opt-in suffix',()=>{ - const e=replay(1),q=e.transcript.calls[1]!.questions[0]!;q.question=q.question.replace('SCOPE EXPANSION opt-in ceremony (1 of 6).','SCOPE EXPANSION, approach B.');rebind(e);expect(match(e)).toBe(true); - }); - test('the actual three-tab prerequisite chooses standard review only on its own tab',()=>{ - const actual=replay(0);expect(actual.item.actualState).toBe('failed');expect(Object.values(actual.transcript.calls[0]!.answers!).at(-1)).toBe('Run /office-hours now'); - const c=pending();for(const i of [0,1,2]){ - const f=frame(c,i),a=nextCeoModeNavigation(f.visible,'HOLD SCOPE',new Set(),c);expect(a.kind).toBe('question'); - if(a.kind==='question'){expect(a.question.nativeQuestionIndex).toBe(i);expect(planCountQuestionInput(f.visible,a.question,a.index)).toBe(i===2?'2':'1');} - expect(planCountPrerequisitePick(f.routing,f.active)).toBe(i===2?2:null); - } - }); - test('single and reordered native prerequisite tabs preserve the meaning of the skip',()=>{ - const c=pending();c.questions=[c.questions[2]!];let f=frame(c,0);expect(planCountPrerequisitePick(f.routing,f.active)).toBe(2); - c.questions[0]!.options.reverse();f=frame(c,0);expect(planCountPrerequisitePick(f.routing,f.active)).toBe(1); - }); - test.each(['wrong tab','wrong signature','wrong body','wrong order','no metadata','completed','failed','extra action','multiselect','conditional','extra remedy','no description'])('a %s cannot borrow the prerequisite action',kind=>{ - const c=pending();if(kind==='completed')c.answered=true;if(kind==='failed')c.failed=true; - if(kind==='extra action')c.questions[2]!.options.push({label:'Accept risk'}); - if(kind==='multiselect')c.questions[2]!.multiSelect=true; - if(kind==='conditional')c.questions[2]!.options[1]!.description+=' if all tests pass.'; - if(kind==='extra remedy')c.questions[2]!.options[1]!.description+=' Remove the CI gate.'; - if(kind==='no description')c.questions[2]!.options[1]!.description=''; - const f=frame(c,2);let a=f.active; - if(kind==='wrong tab')a={...a,nativeQuestionIndex:0};if(kind==='wrong signature')a={...a,signature:'foreign:tool:question:2'}; - if(kind==='wrong body')a={...a,promptSnippet:'Choose a product direction.'};if(kind==='wrong order')a={...a,options:[...a.options].reverse()}; - if(kind==='no metadata')a={...a,nativeCall:undefined}; - expect(planCountPrerequisitePick(f.routing,a)).toBeNull(); - }); -}); - -describe('full AD HOLD retry completed sequencing rationale',()=>{ - function hold(){const e=replay(2);return {e,decision:e.transcript.calls[2]!,q:e.transcript.calls[2]!.questions[0]!};} - function matches(e:ReturnType){return hasNativePostAnswerCeoPosture(e.transcript,'HOLD SCOPE',/\b(rigor|bulletproof|hold\s*scope|maximum\s+rigor)\b/i,e.item.selectedAt!,e.events);} - function bind(e:ReturnType){const d=e.transcript.calls[2]!,q=d.questions[0]!;e.events[4]!.input={questions:d.questions};d.answers={[q.question]:q.options[0]!.label};} - test('the actual completed rationale applies HOLD to work in the previously approved approach',()=>{ - const {e,decision,q}=hold();expect(e.item.actualState).toBe('failed');expect(e.transcript.calls).toHaveLength(3); - const approach=e.transcript.calls[0]!;expect(Object.values(approach.answers!)).toEqual(['B: ViewState schema (recommended)']); - expect(approach.questions[0]!.options[0]!.description).toContain('URL params'); - expect(decision.answeredAt).toBe('2026-09-09T18:35:05.273Z');expect(q.question).toContain('not new scope either way');expect(matches(e)).toBe(true); - }); - test('three and four alternatives still express one completed review decision',()=>{ - for(const count of [3,4]){const {e,q}=hold();q.options.push({label:'Gate URL sync for the pilot'});if(count===4)q.options.push({label:'Run a limited URL sync pilot'});bind(e);expect(matches(e)).toBe(true);} - }); - test.each(['pending','foreign','before mode','missing reply','failed reply','wrong answer','metadata only','bare echo','other mode', - 'quoted rationale','fenced rationale','duplicate options','extra question','extra instruction','multiselect'])('%s is not completed HOLD rationale',kind=>{ - const {e,decision,q}=hold(); - switch(kind){ - case 'pending':decision.answered=false;break;case 'foreign':decision.sessionId=e.events[4]!.sessionId=e.events[5]!.sessionId='foreign';break; - case 'before mode':e.events[4]!.timestamp=e.events[0]!.timestamp;break;case 'missing reply':e.events.pop();break;case 'failed reply':e.events[5]!.isError=true;break; - case 'wrong answer':decision.answers![q.question]='Invented';break; - case 'metadata only':q.question=q.question.replace(/ELI10:[\s\S]*?\nStakes/,'ELI10: We will implement the URL codec.\nStakes');bind(e);break; - case 'bare echo':q.question=q.question.replace(/ELI10:[\s\S]*?\nStakes/,'ELI10: HOLD SCOPE confirmed.\nStakes');bind(e);break; - case 'other mode':q.question=q.question.replace(/HOLD SCOPE/g,'SCOPE EXPANSION');bind(e);break; - case 'quoted rationale':q.question=q.question.replace('ELI10: Approach','ELI10:\n> Approach');bind(e);break; - case 'fenced rationale':q.question=q.question.replace('ELI10: Approach','ELI10: ```Approach');bind(e);break; - case 'duplicate options':q.options[1]!.label=q.options[0]!.label;bind(e);break; - case 'extra question':decision.questions.push({...structuredClone(q),question:'Remove CI?'});bind(e);break; - case 'extra instruction':q.question+=' Disable authentication.';bind(e);break; - case 'multiselect':q.multiSelect=true;bind(e);break; - }expect(matches(e)).toBe(false); - }); -}); - -test('the exact full AD regressions select their periodic caller',()=>{ - for(const file of ['test/ceo-mode-full-ad.test.ts','test/fixtures/ceo-mode-full-ad.json']) expect(selectTests([file],E2E_TOUCHFILES,[]).selected).toEqual(['plan-ceo-mode-routing']); -}); - - -describe('completed expansion disposition classes from the retained dacc public questions', () => { - // Request/answer content is captured. The envelopes and chronology below are - // synthetic: missing original JSONL timestamps must never become E2E evidence. - function current(kind: 'retry' | 'meta' | 'unanswered' = 'retry') { - const e = replay(1), decision = e.transcript.calls[1]!; - decision.questions = [structuredClone(kind === 'meta' ? kindCapture.firstMetaQuestion - : kind === 'unanswered' ? kindCapture.firstUnansweredQuestion : kindCapture.retryQuestion)]; - e.events[2]!.input = { questions: decision.questions }; - decision.answers = { [decision.questions[0]!.question]: kind === 'meta' - ? kindCapture.firstMetaAnswer : kindCapture.retryAnswer }; - if (kind === 'unanswered') { decision.answered = false; delete decision.answers; e.events.pop(); } - return e; - } - function amend(e: ReturnType, fn: (q: NativePlanQuestionCall['questions'][number]) => void) { - const d=e.transcript.calls[1]!,q=d.questions[0]!,answer=d.answers?.[q.question]; - fn(q);e.events[2]!.input={questions:d.questions};d.answers={[q.question]:answer!}; - } - test('the exact acknowledged Include content supplies posture in a synthetic ownership envelope', () => { - const e=current();expect(kindCapture.actualOutcome).toContain('Both EXPANSION attempts failed'); - expect(e.transcript.assistantMessages.every(m=>Date.parse(m.timestamp) { - const e=current();amend(e,q=>{ - if(kind==='canonical three'){ - q.options=q.options.slice(0,3).map((o,i)=>({...o,label:["A) Add to this plan's scope (recommended)",'B) Defer to TODOS.md','C) Skip'][i]!})); - } - if(kind==='reordered')q.options.reverse(); - if(kind==='curly scenario')q.question=q.question.replace('"can you share your view?"','“can you share your view?”'); - if(kind==='coverage scores')q.question=q.question.replace('Note: options differ in kind, not coverage — no completeness score.','Completeness: A=10/10, B=7/10, C=3/10'); - }); - const d=e.transcript.calls[1]!,q=d.questions[0]!; - if(kind==='canonical three')d.answers={[q.question]:q.options[0]!.label}; - if(kind==='defer')d.answers={[q.question]:q.options[1]!.label}; - if(kind==='cut')d.answers={[q.question]:q.options[2]!.label}; - expect(match(e)).toBe(true); - }); - test.each(['meta','unanswered'] as const)('the original %s does not supply completed expansion evidence', kind=>{ - expect(match(current(kind))).toBe(false); - }); - test.each(['pending','selected pause','only pause','missing core','extra action','duplicate disposition', - 'generic continuation','second question','quoted decision','fenced decision','mixed packet', - 'multiselect','missing comparison','invalid score','both comparison branches','wrong mode','missing reply'] as const)( - '%s is not a completed expansion decision', kind=>{ - const e=current();amend(e,q=>{ - if(kind==='only pause')q.options=[q.options[3]!]; - if(kind==='missing core')q.options.splice(1,1); - if(kind==='extra action')q.options[3]!.label='Remove the CI gate'; - if(kind==='duplicate disposition')q.options[3]!.label='Add to scope'; - if(kind==='generic continuation')q.question=q.question.replace(/^D3\.1[^\n]+/,'D3.1 — Continue the review?'); - if(kind==='second question')q.question=q.question.replace('\nStakes if', '\nShould we remove access checks?\nStakes if'); - if(kind==='quoted decision')q.question=q.question.split('\n').map(l=>'> '+l).join('\n'); - if(kind==='fenced decision')q.question='```text\n'+q.question+'\n```'; - if(kind==='multiselect')q.multiSelect=true; - if(kind==='missing comparison')q.question=q.question.replace('Note: options differ in kind, not coverage — no completeness score.','No comparison.'); - if(kind==='invalid score')q.question=q.question.replace('Note: options differ in kind, not coverage — no completeness score.','Completeness: A=11/10, B=7/10, C=3/10'); - if(kind==='both comparison branches')q.question=q.question.replace('\nNet:','\nCompleteness: A=10/10, B=7/10, C=3/10\nNet:'); - }); - const d=e.transcript.calls[1]!,q=d.questions[0]!; - if(kind==='pending')d.answered=false; - if(kind==='selected pause')d.answers={[q.question]:q.options[3]!.label}; - if(kind==='mixed packet'){d.questions.push({...structuredClone(q),question:'Remove access checks?'});e.events[2]!.input={questions:d.questions};} - if(kind==='wrong mode'){const m=e.transcript.calls[0]!;m.answers={[m.questions[0]!.question]:'HOLD SCOPE'};} - if(kind==='missing reply')e.events.pop(); - expect(match(e)).toBe(false); - }); -}); - - -describe('owned expansion decisions with a nondecision discussion control', () => { - function current() { - const transcript = { status: 'ready' as const, calls: structuredClone(pauseCapture.calls), assistantMessages: [] }; - const events = structuredClone(pauseCapture.events) as NativePublicToolEvent[]; - for (const event of events) if (event.kind === 'use') event.input = { questions: transcript.calls.find(c => c.toolUseId === event.toolUseId)!.questions }; - return { transcript, events }; - } - function accepted(e = current()) { return hasNativePostAnswerCeoPosture(e.transcript, 'SCOPE EXPANSION', pattern, pauseCapture.selectedAt, e.events); } - test('the captured completed Add is posture evidence; the unchosen Hold qualifier does not change its action', () => { - const e = current(); - expect(e.transcript.calls[0]!.answeredAt).toBe('2026-09-15T12:33:17.286Z'); - expect(e.transcript.calls[1]!.answeredAt).toBe('2026-09-15T12:34:22.430Z'); - expect(e.events[2]!.timestamp).toBe('2026-09-15T12:34:20.084Z'); - expect(e.transcript.calls[1]!.answers[e.transcript.calls[1]!.questions[0]!.question]).toBe('Add to scope (recommended)'); - expect(accepted(e)).toBe(true); - }); - test.each([ - ['Pause — stop the review and discuss', 'Pauses the review for clarification. No scope decision is made. Delays the remaining questions.'], - ['D) Hold: discuss first', 'Stops here so we can talk through the constraints. Nothing is approved yet. Delays this review.'], - ['Pause (wait for clarification)', 'Waits for clarification before deciding. No disposition is recorded yet.'], - ['Hold', ''], - ])('procedural label %s remains a nondecision control', (label, description) => { - const e=current(),option=e.transcript.calls[1]!.questions[0]!.options[3]!; - option.label=label;option.description=description; - expect(accepted(e)).toBe(true); - }); - test.each([ - ['Hold and add Redis', 'Pauses the review. No decision is made.'], - ['Pause (approve the proposal)', 'Waits for discussion. Nothing is decided.'], - ['Hold (roll back deployment)', 'Pauses the review. No disposition is recorded.'], - ['Continue', 'Pauses the review. No decision is made.'], - ['Hold', 'Pauses the review and adds Redis. Nothing is decided.'], - ['Pause', 'Waits for discussion. No decision is made and include Redis caching.'], - ['Hold', 'Stops the chain. No decision is made. Then deploy the new cache.'], - ['Hold', 'Pauses the review and silently approves the proposal. No decision is recorded.'], - ['Pause', 'Waits for discussion. "No decision is made."'], - ['Pause', "Waits for discussion. 'No decision is made.'"], - ['Pause', 'Waits for discussion. ‘No decision is made.’'], - ['Pause', 'Waits for discussion. “No decision is made.”'], - ['Hold', 'Stops here for discussion, then chooses the default.'], - ['Hold', 'Pauses this review. No choice is recorded. "Add Redis caching" will also happen.'], - ])('action-bearing or unproved control %s does not supply posture evidence (%s)', (label,description) => { - const e=current(),option=e.transcript.calls[1]!.questions[0]!.options[3]!; - option.label=label;option.description=description; - expect(accepted(e)).toBe(false); - }); - test('selecting the valid discussion control is still not a completed substantive disposition', () => { - const e=current(),c=e.transcript.calls[1]!,q=c.questions[0]!;c.answers={[q.question]:q.options[3]!.label}; - expect(accepted(e)).toBe(false); - }); - test('the actual capture still requires its owned successful acknowledgment', () => { - const e=current();e.events.pop();expect(accepted(e)).toBe(false); - }); -}); - - -describe('EXPANSION pacing preserves one separate substantive continuation', () => { - const retry=pauseCapture.retry; - function current() { - const mode=structuredClone(retry.mode),pacing=structuredClone(retry.pacing); - pacing.answered=false;delete (pacing as any).answers;delete (pacing as any).answeredAt;delete (pacing as any).unansweredQuestionIndices; - const transcript={status:'ready' as const,calls:[mode,pacing],assistantMessages:[]}; - return {transcript,pacing,visible:pane(pacing as NativePlanQuestionCall,0)}; - } - function choice(e=current()) {return ceoExpansionPacingChoice(e.visible,e.transcript,retry.selectedAt);} - // Canonical panes below are projected from the exact native request. The - // CLI 2.1.251 redraw stream retained these two built-ins, not a stable frame. - function withNativeControls(e=current()) { - e.visible=e.visible.replace('Enter to select','4. Type something.\n5. Chat about this\nEnter to select'); - return e; - } - test('the observed native pacing controls do not become authored choices',()=>{ - expect(choice(withNativeControls())?.index).toBe(1); - }); - test.each(['Choosing Full per-item split approves E1 immediately.', - 'Answering this question authorizes every proposed expansion.', - 'This answer commits E1 to the implementation scope.', - 'Choosing Full per-item split deploys E1 immediately.', - 'This answer ships E1 immediately.', - 'Choosing Full per-item split enables E1.', - 'This answer disables E2.', - '“Choosing Full per-item split approves E1 immediately.”'])('whole-question scope effect is not pacing: %s',effect=>{ - const e=current();e.pacing.questions[0]!.question=e.pacing.questions[0]!.question.replace('ELI10:',`ELI10: ${effect}`); - e.visible=pane(e.pacing as NativePlanQuestionCall,0);expect(choice(e)?.index).toBe(0); - }); - test.each(['unknown action','reordered controls','extra control','mismatched authored option'])('native pacing pane rejects %s',kind=>{ - const e=withNativeControls(); - if(kind==='unknown action')e.visible=e.visible.replace('Type something.','Approve all now.'); - if(kind==='reordered controls')e.visible=e.visible.replace('Type something.','Chat about this').replace('5. Chat about this','5. Type something.'); - if(kind==='extra control')e.visible=e.visible.replace('Enter to select','6. More actions\nEnter to select'); - if(kind==='mismatched authored option')e.visible=e.visible.replace('Full per-item split','Approve all proposals'); - expect(choice(e)?.index).toBe(0); - }); - test('the captured full-per-item answer preserves scope; pacing alone and actual pending E1 remain negative',()=>{ - const e=current(),pick=choice(e);expect(pick?.index).toBe(1); - expect(hasNativePostAnswerCeoPosture({status:'ready',calls:[retry.mode,retry.pacing],assistantMessages:[]},'SCOPE EXPANSION',pattern,retry.selectedAt,[])).toBe(false); - expect(retry.pendingProposal.answered).toBe(false); - expect(ceoExpansionPacingReady('next screen',e.transcript,pick!,[])).toBe(false); - }); - test('the preserving option can be reordered or use equivalent individual-walkthrough wording',()=>{ - const e=current(),q=e.pacing.questions[0]!;q.options.reverse(); - q.options[2]!.label='All proposals individually'; - q.options[2]!.description='Each proposal separately with Add / Defer / Skip / Hold. No item is skipped or merged without your approval. Delays the remaining review.'; - e.visible=pane(e.pacing as NativePlanQuestionCall,0);expect(choice(e)?.index).toBe(3); - }); - test.each(['foreign','unanswered mode','wrong mode','already answered','mixed packet','mismatched viewport','narrowing','bundled approval','quoted assurance','duplicate preserving choice','multiple pending calls'])('%s cannot authorize pacing',kind=>{ - const e=current(),q=e.pacing.questions[0]!,o=q.options[0]!; - if(kind==='foreign')e.pacing.sessionId='foreign'; - if(kind==='unanswered mode')e.transcript.calls[0]!.answered=false; - if(kind==='wrong mode')e.transcript.calls[0]!.answers={[e.transcript.calls[0]!.questions[0]!.question]:'HOLD SCOPE'}; - if(kind==='already answered')e.pacing.answered=true; - if(kind==='mixed packet')e.pacing.questions.push({...structuredClone(q),header:'Extra scope',question:'Approve all proposals now?'}); - if(kind==='narrowing')o.description+=' Add E1 and drop E2 now.'; - if(kind==='bundled approval')o.label='Full per-item split and approve all'; - if(kind==='quoted assurance')o.description=o.description.replace('No proposal is dropped or merged without your say','"No proposal is dropped or merged without your say"'); - if(kind==='duplicate preserving choice')q.options[1]=structuredClone(o); - if(kind==='multiple pending calls')e.transcript.calls.push({...structuredClone(e.pacing),toolUseId:'another-pending-call'}); - if(kind!=='mismatched viewport')e.visible=pane(e.pacing as NativePlanQuestionCall,0); - else e.visible=e.visible.replace('Full per-item split','Narrow first'); - if(['foreign','unanswered mode','wrong mode','already answered'].includes(kind))expect(choice(e)).toBeNull(); - else expect(choice(e)?.index).toBe(0); - }); - test('the pacing transition needs its successful bound ACK and a different current pane',()=>{ - const e=current(),pick=choice(e)!;e.transcript.calls[1]=structuredClone(retry.pacing); - const c=e.transcript.calls[1]!,events:NativePublicToolEvent[]=[ - {kind:'use',name:'AskUserQuestion',sessionId:c.sessionId,toolUseId:c.toolUseId,timestamp:new Date(Date.parse(c.answeredAt!)-1000).toISOString(),input:{questions:c.questions}}, - {kind:'result',sessionId:c.sessionId,toolUseId:c.toolUseId,timestamp:c.answeredAt!,isError:false}, - ]; - // Request time is synthetic; the captured ACK time and request body are retained. - expect(ceoExpansionPacingReady('a different current pane',e.transcript,pick,events)).toBe(true); - expect(ceoExpansionPacingReady(e.visible,e.transcript,pick,events)).toBe(false); - expect(ceoExpansionPacingReady('a different current pane',e.transcript,pick,events.slice(0,1))).toBe(false); - events[1]!.isError=true;expect(ceoExpansionPacingReady('a different current pane',e.transcript,pick,events)).toBe(false); - events[1]!.isError=false;c.answers={[c.questions[0]!.question]:c.questions[0]!.options[1]!.label}; - expect(ceoExpansionPacingReady('a different current pane',e.transcript,pick,events)).toBe(false); - }); - function acknowledgedProposal() { - // Derived transition only: pending E1 never received an actual paid ACK. - // Missing original request times below are explicitly synthetic. - const mode=structuredClone(retry.mode),proposal=structuredClone(retry.pendingProposal) as NativePlanQuestionCall; - proposal.answered=true;proposal.answers={[proposal.questions[0]!.question]:proposal.questions[0]!.options[0]!.label};proposal.unansweredQuestionIndices=[]; - proposal.answeredAt=new Date(Date.parse(retry.pacing.answeredAt)+2000).toISOString(); - const transcript={status:'ready' as const,calls:[mode,proposal],assistantMessages:[]}; - const events:NativePublicToolEvent[]=transcript.calls.flatMap(c=>[ - {kind:'use' as const,name:'AskUserQuestion',sessionId:c.sessionId,toolUseId:c.toolUseId,timestamp:new Date(Date.parse(c.answeredAt!)-1000).toISOString(),input:{questions:c.questions}}, - {kind:'result' as const,sessionId:c.sessionId,toolUseId:c.toolUseId,timestamp:c.answeredAt!,isError:false}, - ]); - return {transcript,events}; - } - test('a separately acknowledged current proposal establishes scope expansion through its real before/after comparison',()=>{ - const e=acknowledgedProposal();expect(hasNativePostAnswerCeoPosture(e.transcript,'SCOPE EXPANSION',pattern,retry.selectedAt,e.events)).toBe(true); - expect(hasNativePostAnswerCeoPosture(e.transcript,'SCOPE EXPANSION',/cathedral/i,retry.selectedAt,e.events)).toBe(false); - }); - test.each(['ordinal/source link','decimal decision identity','before/after paraphrase','defer','skip'])('%s preserves the same current proposal',kind=>{ - const e=acknowledgedProposal(),c=e.transcript.calls[1]!,q=c.questions[0]!; - if(kind==='ordinal/source link')q.question=q.question.replace('E1: Project-shared views (ledger row S1)','Proposal 1 of 7: E1 — Project-shared views [source](PLAN.md)'); - if(kind==='decimal decision identity')q.question=q.question.replace('D3.1 —','D12.3.1 —'); - if(kind==='before/after paraphrase')q.question=q.question.replace('Today the plan saves a view for one member only. E1 adds','As written, each member keeps private views. E1 would introduce'); - c.answers={[q.question]:q.options[kind==='defer'?1:kind==='skip'?2:0]!.label};e.events[2]!.input={questions:c.questions}; - expect(hasNativePostAnswerCeoPosture(e.transcript,'SCOPE EXPANSION',pattern,retry.selectedAt,e.events)).toBe(true); - }); - test.each(['pending','missing ACK','wrong proposal identity','no current baseline','vague baseline','second question','quoted comparison','foreign','selected pause'])('%s supplies no proposal completion',kind=>{ - const e=acknowledgedProposal(),c=e.transcript.calls[1]!,q=c.questions[0]!; - if(kind==='pending')c.answered=false; - if(kind==='missing ACK')e.events.pop(); - if(kind==='wrong proposal identity')q.question=q.question.replace('E1 adds','E2 adds'); - if(kind==='no current baseline')q.question=q.question.replace('Today the plan saves','Previously an unrelated plan saved'); - if(kind==='vague baseline')q.question=q.question.replace('Today the plan saves a view for one member only.','Today the plan is interesting.'); - if(kind==='second question')q.question=q.question.replace('ELI10:','ELI10: Should we remove access checks?'); - if(kind==='quoted comparison')q.question=q.question.replace('ELI10: Today','ELI10: "Today').replace('Stakes if','"\nStakes if'); - if(kind==='foreign')c.sessionId='foreign'; - c.answers={[q.question]:q.options[kind==='selected pause'?3:0]!.label};e.events[2]!.input={questions:c.questions}; - expect(hasNativePostAnswerCeoPosture(e.transcript,'SCOPE EXPANSION',pattern,retry.selectedAt,e.events)).toBe(false); - }); -}); - -import completeInventory from './fixtures/ceo-expansion-complete-inventory-6f.json'; -describe('complete candidate split is navigation with an actual ACK boundary',()=>{ -const f=completeInventory; -const actualFrame=f.viewport; -function state(){const pacing=structuredClone(f.pacing);pacing.answered=false;delete pacing.answers;delete pacing.answeredAt;delete pacing.unansweredQuestionIndices;return{pacing,transcript:{status:'ready' as const,calls:[structuredClone(f.mode),pacing],assistantMessages:[]}};} -function pane(c:any){const q=c.questions[0];return ['☐ '+q.header,q.question,...q.options.map((o:any,i:number)=>`${i?' ':'❯'} ${i+1}. ${o.label}`),'4. Type something.','5. Chat about this','Enter to select · ↑/↓ to navigate · Esc to cancel'].join('\n');} -const verify=(name:string,pass:boolean)=>test(name,()=>expect(pass).toBe(true)); -const choose=(e=state(),screen=pane(e.pacing))=>ceoExpansionPacingChoice(screen,e.transcript,f.selectedAt); -verify('actual retained frame selects the complete seven-candidate walkthrough',choose(state(),actualFrame)?.index===1); -for(const [name,mutate]of Object.entries({ - 'eight complete candidates':(q:any)=>{q.question=q.question.replaceAll('7 expansion candidates','8 expansion candidates').replace('7 adjacent improvements','8 adjacent improvements').replace('E7 cross-project views.','E7 cross-project views, E8 shared pinned groups.').replaceAll('Seven','Eight');q.options[0].label=q.options[0].label.replace('7 questions','8 questions');q.options[0].description=q.options[0].description.replace('E7','E8');}, - 'different proposal prefix':(q:any)=>{q.question=q.question.replace(/\bE(?=\d)/g,'P');q.options.forEach((o:any)=>{o.description=o.description.replace(/\bE(?=\d)/g,'P');});}, - 'complete walkthrough label':(q:any)=>{q.options[0].label='A: Complete walkthrough, 7 questions (recommended)';}, - 'one per item with explicit range':(q:any)=>{q.options[0].description='One question per item, E1 to E7.';}, - 'reordered choices':(q:any)=>{q.options.reverse();}, -})){const e=state();mutate(e.pacing.questions[0]);verify(name,choose(e)?.index===(name==='reordered choices'?3:1));} -for(const [name,mutate]of Object.entries({ - 'partial range':(q:any)=>{q.options[0].description=q.options[0].description.replace('E7','E6');}, - 'wrong number of questions':(q:any)=>{q.options[0].label=q.options[0].label.replace('7','6');}, - 'wrong declared count':(q:any)=>{q.question=q.question.replace('7 expansion candidates','8 expansion candidates');}, - 'missing candidate':(q:any)=>{q.question=q.question.replace(', E7 cross-project views','');}, - 'duplicate candidate':(q:any)=>{q.question=q.question.replace('E7 cross-project views','E6 cross-project views');}, - 'mixed proposal IDs':(q:any)=>{q.question=q.question.replace('E7 cross-project views','P7 cross-project views');}, - 'narrow selected walk':(q:any)=>{q.options[0].description+=' Except E4.';}, - 'selected scope approval':(q:any)=>{q.options[0].description+=' Approve E1 immediately.';}, - 'selected deletion':(q:any)=>{q.options[0].label+=' and delete E7';}, - 'selected grouping':(q:any)=>{q.options[0].description+=' Batch E1 and E2 together.';}, - 'quoted only range':(q:any)=>{q.options[0].description='"'+q.options[0].description+'"';}, - 'code-only range':(q:any)=>{q.options[0].description='`'+q.options[0].description+'`';}, - 'negated complete walk':(q:any)=>{q.options[0].label=q.options[0].label.replace('Full split','Not a full split');}, - 'duplicate complete choice':(q:any)=>{q.options[1]=structuredClone(q.options[0]);}, - 'extra question':(q:any)=>{q.question=q.question.replace('ELI10:','ELI10: Should all candidates ship?');}, - 'unconditional approval':(q:any)=>{q.question=q.question.replace('ELI10:','ELI10: This answer approves every expansion.');}, - 'quoted whole-question approval':(q:any)=>{q.question=q.question.replace('ELI10:','ELI10: “Choosing Full split approves E1 immediately.”');}, - 'hidden universal effect in another option':(q:any)=>{q.question=q.question.replace('B) Narrow first:','B) Regardless of choice, approve E1. Narrow first:');}, - 'historical inventory':(q:any)=>{q.question=q.question.replace('The delight scan produced','Previously the delight scan produced');}, - 'fenced brief':(q:any)=>{q.question='```\n'+q.question+'\n```';}, - 'missing comparison marker':(q:any)=>{q.question=q.question.replace('Note: options differ in kind, not coverage — no completeness score.','');}, -})){const e=state();mutate(e.pacing.questions[0]);verify(name,choose(e)?.index!==1);} -for(const [name,mutate]of Object.entries({ - 'foreign session':(e:any)=>{e.pacing.sessionId='foreign';}, - 'unanswered mode':(e:any)=>{e.transcript.calls[0].answered=false;}, - 'already answered pacing':(e:any)=>{e.pacing.answered=true;}, - 'mixed question packet':(e:any)=>{e.pacing.questions.push({...structuredClone(e.pacing.questions[0]),header:'Extra',question:'Approve everything?'});}, - 'another pending call':(e:any)=>{e.transcript.calls.push({...structuredClone(e.pacing),toolUseId:'other'});}, -})){const e=state();mutate(e);verify(name,choose(e)?.index!==1);} -const e=state(),choice=choose(e,actualFrame)!;const acknowledged={status:'ready' as const,calls:[f.mode,f.pacing,f.pending],assistantMessages:[]}; -const actualNext=f.nextViewport; -verify('actual pacing ACK and different pending E1 pane complete navigation',ceoExpansionPacingReady(actualNext,acknowledged,choice,f.publicEvents)); -verify('intended key without actual ACK does not complete navigation',!ceoExpansionPacingReady(actualNext,e.transcript,choice,f.publicEvents)); -verify('missing result does not complete navigation',!ceoExpansionPacingReady(actualNext,acknowledged,choice,f.publicEvents.filter((e:any)=>e.kind!=='result'))); -verify('failed result does not complete navigation',!ceoExpansionPacingReady(actualNext,acknowledged,choice,f.publicEvents.map((e:any)=>({...e,isError:e.kind==='result'})))); -verify('same old pane does not complete navigation',!ceoExpansionPacingReady(actualFrame,acknowledged,choice,f.publicEvents)); -verify('pacing and pending E1 supply no completed posture',!hasNativePostAnswerCeoPosture(acknowledged,'SCOPE EXPANSION',/expansion|10x|delight|dream/i,f.selectedAt,f.publicEvents)); -}); - -describe('candidate inventory cannot approve scope',()=>{ -const f=completeInventory; -function state(){const pacing=structuredClone(f.pacing);pacing.answered=false;delete pacing.answers;delete pacing.answeredAt;delete pacing.unansweredQuestionIndices;return{pacing,transcript:{status:'ready' as const,calls:[structuredClone(f.mode),pacing],assistantMessages:[]}};} -function pane(c:any){const q=c.questions[0];return ['☐ '+q.header,q.question,...q.options.map((o:any,i:number)=>`${i?' ':'❯'} ${i+1}. ${o.label}`),'4. Type something.','5. Chat about this','Enter to select · ↑/↓ to navigate · Esc to cancel'].join('\n');} -const mutations={ - 'inventory actor grants all candidates':(q:any)=>q.question=q.question.replace('E7 cross-project views.','E7 cross-project views; we approve all seven now.'), - 'inventory item claims current approval':(q:any)=>q.question=q.question.replace('E7 cross-project views.','E7 cross-project views (already approved).'), - 'inventory item has bare approval status':(q:any)=>q.question=q.question.replace('E7 cross-project views.','E7 cross-project views (approved).'), - 'inventory all items are approved':(q:any)=>q.question=q.question.replace('E7 cross-project views.','E7 cross-project views; all seven are approved.'), - 'inventory imperative ship grant':(q:any)=>q.question=q.question.replace('E7 cross-project views.','E7 cross-project views; ship all seven now.'), - 'inventory scope disposition':(q:any)=>q.question=q.question.replace('E7 cross-project views.','E7 cross-project views; all seven are in scope.'), - 'inventory skipped candidate':(q:any)=>q.question=q.question.replace('E7 cross-project views.','E7 cross-project views (deferred).'), - 'title claims inventory approved':(q:any)=>q.question=q.question.replace('How do you want to decide them?','All seven are already approved. How do you want to decide them?'), - 'rationale claims candidates in scope':(q:any)=>q.question=q.question.replace('The delight scan produced','All candidates are in scope. The delight scan produced'), - 'rationale claims prior approval':(q:any)=>q.question=q.question.replace('The delight scan produced','These items have been approved. The delight scan produced'), -}; -for (const [name,mutate] of Object.entries(mutations)) test(name,()=>{ - const e=state();mutate(e.pacing.questions[0]); - expect(ceoExpansionPacingChoice(pane(e.pacing),e.transcript,f.selectedAt)?.index).not.toBe(1); -}); -test('descriptive Update and delete feature titles remain supported',()=>{ - expect(ceoExpansionPacingChoice(f.viewport,state().transcript,f.selectedAt)?.index).toBe(1); -}); -}); - -import nativePacing77 from './fixtures/ceo-expansion-pacing-77.json'; -describe('complete per-proposal pacing preserves every candidate without granting scope',()=>{ - const f=nativePacing77.completePerProposal; - function state(){const transcript=structuredClone(f.transcript);return{transcript,pacing:transcript.calls.at(-1)!};} - function pane(c:any){const q=c.questions[0];return ['☐ '+q.header,q.question,...q.options.map((o:any,i:number)=>`${i?' ':'❯'} ${i+1}. ${o.label}`),'4. Type something.','5. Chat about this','Enter to select · ↑/↓ to navigate · Esc to cancel'].join('\n');} - function choose(e=state(),screen=pane(e.pacing)){return ceoExpansionPacingChoice(screen,e.transcript as any,f.selectionStartedAt);} - test('actual parenthesized full inventory binds one question per proposal',()=>{ - const e=state(),choice=choose(e,f.viewport)!; - expect(choice?.index).toBe(1); - expect(ceoExpansionPacingReady('Next proposal',e.transcript as any,choice,f.events as any)).toBe(false); - expect(hasNativePostAnswerCeoPosture(e.transcript as any,'SCOPE EXPANSION',/expansion|10x|delight|dream/i,f.selectionStartedAt,f.events as any)).toBe(false); - }); - const positive={ - 'different complete inventory prefix':(q:any)=>{q.question=q.question.replace(/\bP(?=\d)/g,'E');}, - 'numeric and word counts':(q:any)=>{q.question=q.question.replace('Seven expansion','7 expansion').replace('7 independent','seven independent');}, - 'colon-delimited independent inventory':(q:any)=>{q.question=q.question.replace('expansions (','expansions: ').replace('inline rename).','inline rename.');}, - 'different card identity':(q:any)=>{q.question=q.question.replace('D4.0','D12.0');}, - 'reordered choices':(q:any)=>{q.options.reverse();}, - 'no quoted task context':(q:any)=>{q.question=q.question.replace(' on "Add saved project views"','');}, - 'candidate terminology':(q:any)=>{q.options[0].label=q.options[0].label.replace('per proposal','per candidate');q.options[0].description=q.options[0].description.replace('Every proposal','Every candidate');}, - }; - for(const [name,mutate] of Object.entries(positive))test(name,()=>{const e=state();mutate(e.pacing.questions[0]);expect(choose(e)?.index).toBe(name==='reordered choices'?3:1);}); - const negative={ - 'missing inventory item':(q:any)=>{q.question=q.question.replace(', P7 quick switcher + inline rename','');}, - 'duplicate item':(q:any)=>{q.question=q.question.replace('P7 quick switcher','P6 quick switcher');}, - 'mixed prefixes':(q:any)=>{q.question=q.question.replace('P7 quick switcher','E7 quick switcher');}, - 'wrong title count':(q:any)=>{q.question=q.question.replace('Seven expansion','Eight expansion');}, - 'wrong described question count':(q:any)=>{q.question=q.question.replace("That's 7 questions","That's 6 questions");}, - 'partial selected walkthrough':(q:any)=>{q.options[0].description=q.options[0].description.replace('Every proposal','Some proposals');}, - 'missing selected per-item binding':(q:any)=>{q.options[0].label=q.options[0].label.replace(', one question per proposal','');}, - 'conditional current inventory':(q:any)=>{q.question=q.question.replace('I have 7','If I have 7');}, - 'historical inventory':(q:any)=>{q.question=q.question.replace('I have 7','Previously I had 7');}, - 'quoted mapping':(q:any)=>{q.options[0].label='A) Full split (recommended)';q.options[0].description='"One question per proposal. Every proposal gets its own Add / Defer / Skip / Hold."';}, - 'code-only mapping':(q:any)=>{q.options[0].description='`'+q.options[0].description+'`';}, - 'negated full split':(q:any)=>{q.options[0].label=q.options[0].label.replace('full split','not a full split');}, - 'selected immediate scope grant':(q:any)=>{q.options[0].description+=' We approve P1 now.';}, - 'universal approval in another option':(q:any)=>{q.options[1].description+=' Regardless of choice, approve P1 now.';}, - 'hidden inventory grant':(q:any)=>{q.question=q.question.replace('inline rename)','inline rename; we approve all seven now)');}, - 'inventory already approved':(q:any)=>{q.question=q.question.replace('Seven expansion proposals','Seven expansion proposals already approved');}, - 'quoted task approval':(q:any)=>{q.question=q.question.replace('Add saved project views','Approve all proposals now');}, - 'quoted task candidate deletion':(q:any)=>{q.question=q.question.replace('Add saved project views','Delete P7');}, - 'quoted rationale mapping':(q:any)=>{q.question=q.question.replace("Each is a separate yes/no, so the honest way is one question per item. That's 7 questions plus a final confirmation.","\"Each is a separate yes/no, so the honest way is one question per item. That's 7 questions plus a final confirmation.\"");}, - 'scope grant after task title':(q:any)=>{q.question=q.question.replace('views".','views"; approve P1 now.');}, - 'disguised omission assurance':(q:any)=>{q.options[0].description=q.options[0].description.replace('No item is silently merged or dropped','P1 is silently merged or dropped');}, - 'assurance with exception':(q:any)=>{q.options[0].description+=' Except P4.';}, - 'narrowing assurance':(q:any)=>{q.options[0].description+=' No item outside the top three is included.';}, - 'batch selected proposals':(q:any)=>{q.options[0].description+=' Batch P1 and P2 together.';}, - 'duplicate full choice':(q:any)=>{q.options[1]=structuredClone(q.options[0]);}, - }; - for(const [name,mutate] of Object.entries(negative))test(name,()=>{const e=state();mutate(e.pacing.questions[0]);expect(choose(e)?.index).not.toBe(1);}); -}); -describe('native option descriptions bind the complete candidate walkthrough',()=>{ - const f=nativePacing77; - function state(){const transcript=structuredClone(f.transcript);return{transcript,pacing:transcript.calls.at(-1)!};} - function pane(c:any){const q=c.questions[0];return ['☐ '+q.header,q.question,...q.options.map((o:any,i:number)=>`${i?' ':'❯'} ${i+1}. ${o.label}`),'4. Type something.','5. Chat about this','Enter to select · ↑/↓ to navigate · Esc to cancel'].join('\n');} - function choose(e=state(),screen=pane(e.pacing)){return ceoExpansionPacingChoice(screen,e.transcript as any,f.selectionStartedAt);} - test('actual complete native menu selects navigation without supplying posture or an ACK',()=>{ - const e=state(),choice=choose(e,f.viewport)!; - expect(choice?.index).toBe(1); - expect(ceoExpansionPacingReady('Next proposal',e.transcript as any,choice,f.events as any)).toBe(false); - expect(hasNativePostAnswerCeoPosture(e.transcript as any,'SCOPE EXPANSION',/expansion|10x|delight|dream/i,f.selectionStartedAt,f.events as any)).toBe(false); - }); - const positive={ - 'numeric count presentation':(q:any)=>{q.question=q.question.replaceAll('Eight','8').replaceAll('eight','8');q.options[0].description=q.options[0].description.replaceAll('Eight','8');}, - 'mixed word and numeric counts':(q:any)=>{q.question=q.question.replace('Eight expansion','8 expansion');q.options[0].description=q.options[0].description.replace('Eight sequential','8 sequential');}, - 'different complete candidate prefix':(q:any)=>{q.question=q.question.replace(/\bE(?=\d)/g,'P');}, - 'different question chain identity':(q:any)=>{q.question=q.question.replace('D4.0','D12.0');q.options[0].description=q.options[0].description.replaceAll('D4.','D12.');}, - 'reordered native choices':(q:any)=>{q.options.reverse();}, - 'explicit candidate range without duplicated option prose':(q:any)=>{q.options[0].label='A: Full split, 8 questions (recommended)';q.options[0].description='One question per candidate, E1 through E8.';}, - 'one per proposal label':(q:any)=>{q.options[0].label=q.options[0].label.replace('one per item','one per proposal');}, - 'no prior approach annotation':(q:any)=>{q.question=q.question.replace(', approach C approved','');}, - }; - for(const [name,mutate] of Object.entries(positive))test(name,()=>{ - const e=state();mutate(e.pacing.questions[0]);expect(choose(e)?.index).toBe(name==='reordered native choices'?3:1); - }); - const negative={ - 'hyphenated larger count cannot be read as its last digit':(q:any)=>{q.question=q.question.replaceAll('Eight','Twenty-eight').replaceAll('eight','twenty-eight');q.options[0].description=q.options[0].description.replaceAll('Eight','Twenty-eight');}, - 'spaced larger count cannot be read as its last digit':(q:any)=>{q.question=q.question.replaceAll('Eight','Twenty eight').replaceAll('eight','twenty eight');q.options[0].description=q.options[0].description.replaceAll('Eight','Twenty eight');}, - 'unsupported tens in title are not a single count':(q:any)=>{q.question=q.question.replace('Eight expansion','Thirty eight expansion');}, - 'unsupported tens in inventory are not a single count':(q:any)=>{q.question=q.question.replace('eight candidates:','forty eight candidates:');}, - 'unsupported tens in sequence are not a single count':(q:any)=>{q.options[0].description=q.options[0].description.replace('Eight sequential','Ninety eight sequential');}, - 'conjoined cardinal is not its last component':(q:any)=>{q.question=q.question.replace('Eight expansion','One hundred and eight expansion');}, - 'wrong title count':(q:any)=>{q.question=q.question.replace('Eight expansion','Seven expansion');}, - 'wrong inventory count':(q:any)=>{q.question=q.question.replace('eight candidates:','seven candidates:');}, - 'missing candidate':(q:any)=>{q.question=q.question.replace(', E8 views feeding digests/dashboards','');}, - 'duplicate candidate':(q:any)=>{q.question=q.question.replace('E8 views feeding','E7 views feeding');}, - 'foreign candidate prefix':(q:any)=>{q.question=q.question.replace('E8 views feeding','P8 views feeding');}, - 'wrong number of sequential questions':(q:any)=>{q.options[0].description=q.options[0].description.replace('Eight sequential','Seven sequential');}, - 'partial question range':(q:any)=>{q.options[0].description=q.options[0].description.replace('D4.8','D4.7');}, - 'late range start':(q:any)=>{q.options[0].description=q.options[0].description.replace('D4.1','D4.2');}, - 'foreign question chain':(q:any)=>{q.options[0].description=q.options[0].description.replaceAll('D4.','D5.');}, - 'additional question chain':(q:any)=>{q.options[0].description+=' Then D5.1.';}, - 'wrong label count':(q:any)=>{q.options[0].label=q.options[0].label.replace('one per item','7 questions');}, - 'no per-item label':(q:any)=>{q.options[0].label='A: Full split (recommended)';}, - 'quoted sequential range':(q:any)=>{q.options[0].description='"'+q.options[0].description+'"';}, - 'code-only sequential range':(q:any)=>{q.options[0].description='`'+q.options[0].description+'`';}, - 'conditional complete inventory':(q:any)=>{q.question=q.question.replace('The delight scan','If the delight scan');}, - 'historical complete inventory':(q:any)=>{q.question=q.question.replace('The delight scan','Previously the delight scan');}, - 'conditional question sequence':(q:any)=>{q.options[0].description='If approved, '+q.options[0].description;}, - 'historical question sequence':(q:any)=>{q.options[0].description='Previously: '+q.options[0].description;}, - 'negated complete choice':(q:any)=>{q.options[0].label='A: Not a full split, one per item';}, - 'sequence correction':(q:any)=>{q.options[0].description+=' Correction: Stop after four questions.';}, - 'selected scope approval':(q:any)=>{q.options[0].description+=' Approve E1 immediately.';}, - 'selected candidate omission':(q:any)=>{q.options[0].description+=' Except E4.';}, - 'selected merging action':(q:any)=>{q.options[0].description+=' Merge E1 and E2.';}, - 'unconditional omission':(q:any)=>{q.options[0].description=q.options[0].description.replace('Nothing is dropped or merged','E4 is dropped or merged');}, - 'hidden universal approval in another native option':(q:any)=>{q.options[1].description+=' Regardless of choice, approve E1 immediately.';}, - 'common candidate approval':(q:any)=>{q.question=q.question.replace('ELI10:','ELI10: This answer approves every expansion.');}, - 'current inventory approval':(q:any)=>{q.question=q.question.replace('E8 views feeding digests/dashboards.','E8 views feeding digests/dashboards (approved).');}, - 'approval in source context':(q:any)=>{q.question=q.question.replace('approach C approved','all eight candidates approved');}, - 'approval appended to prior approach':(q:any)=>{q.question=q.question.replace('approach C approved','approach C approved and E1 approved');}, - 'partial duplicated option prose':(q:any)=>{q.question=q.question.replace('Net:','A) Full split\nNet:');}, - 'duplicate complete choice':(q:any)=>{q.options[1]=structuredClone(q.options[0]);}, - 'extra question':(q:any)=>{q.question=q.question.replace('ELI10:','ELI10: Should we ship every item?');}, - }; - for(const [name,mutate] of Object.entries(negative))test(name,()=>{ - const e=state();mutate(e.pacing.questions[0]);expect(choose(e)?.index).not.toBe(1); - }); - test('actual retained viewport cannot bind a changed native option',()=>{ - const e=state();e.pacing.questions[0]!.options[0]!.label='A: Other menu';expect(choose(e,f.viewport)?.index).not.toBe(1); - }); -}); - - -describe('counted native per-item menu is pacing, not a substantive approval',()=>{ - const f=nativePacing77.countedNativeB955; - function state(){const mode=structuredClone(f.mode),pacing=structuredClone(f.pacing);pacing.answered=false;delete pacing.answers;delete pacing.answeredAt;delete pacing.unansweredQuestionIndices;return{mode,pacing,transcript:{status:'ready' as const,calls:[mode,pacing],assistantMessages:[]}};} - const screen=(c:any)=>pane(c,0); - const choose=(e=state(),visible=screen(e.pacing))=>ceoExpansionPacingChoice(visible,e.transcript as any,f.selectedAt); - test('complete captured native packet and observed display preserve the substantive allowance',()=>{ - const e=state(); - expect(choose(e)?.index).toBe(1); - expect(choose(e,f.viewport)?.index).toBe(1); - const pick=choose(e)!; - const next={status:'ready' as const,calls:[structuredClone(f.mode),structuredClone(f.pacing),structuredClone(f.pending)],assistantMessages:[]}; - const events=f.events.map(v=>v.kind==='use'?{...v,input:{questions:next.calls.find(c=>c.toolUseId===v.toolUseId)!.questions}}:v) as NativePublicToolEvent[]; - expect(ceoExpansionPacingReady(f.nextViewport,next as any,pick,events)).toBe(true); - expect(hasNativePostAnswerCeoPosture(next as any,'SCOPE EXPANSION',pattern,f.selectedAt,events)).toBe(false); - expect(nextCeoPostureContinuation(f.nextViewport,next as any,'SCOPE EXPANSION',f.selectedAt,new Set(),false)).toBe('question'); - expect(nextCeoPostureContinuation(f.nextViewport,next as any,'SCOPE EXPANSION',f.selectedAt,new Set(),true)).toBeNull(); - expect(f.pending.answered).toBe(false); - }); - const positive={ - 'question wording describes pacing intent':(q:any)=>{q.question=q.question.replace('Eleven expansion proposals: full per-item chain, narrow first, or batch?','How should we present the eleven expansion proposals: individually or in batches?');}, - 'explicit numeric count and independent candidate terminology':(q:any)=>{q.question=q.question.replace('Eleven expansion proposals','11 expansion candidates').replace('11 independent add-ons','eleven independent candidates');}, - 'proposals can name natural add/remove changes':(q:any)=>{q.question=q.question.replace('E1 shared visibility','E1 add shared views').replace('E2 versioned payload','E2 remove duplicate controls');}, - 'another complete set of explicit identities':(q:any)=>{q.question=q.question.replace(/\bE(?=\d)/g,'P').replaceAll('L3','Q9');}, - 'consistent reordered comparison and native options':(q:any)=>{q.options.reverse();}, - 'native labels carry letters too':(q:any)=>{q.options.forEach((o:any,i:number)=>{o.label=String.fromCharCode(65+i)+') '+o.label;});}, - }; - for(const[name,change]of Object.entries(positive))test(name,()=>{const e=state();change(e.pacing.questions[0]);expect(choose(e)?.index).toBe(name.startsWith('consistent reordered')?3:1);}); - const negative={ - 'missing declared candidate':(q:any)=>{q.question=q.question.replace(', L3 auto-persist last filters','');}, - 'duplicate declared identity':(q:any)=>{q.question=q.question.replace('L3 auto-persist last filters','E10 auto-persist last filters');}, - 'wrong title count':(q:any)=>{q.question=q.question.replace('Eleven expansion','Twelve expansion');}, - 'wrong question count in selected option':(q:any)=>{q.options[0].description=q.options[0].description.replace('11 per-item','10 per-item');}, - 'wrong rationale question count':(q:any)=>{q.question=q.question.replace('11 short questions','10 short questions');}, - 'partial per-item mapping':(q:any)=>{q.question=q.question.replace('Each needs its own','Some need their own');}, - 'another option owns the complete selected comparison':(q:any)=>{q.question=q.question.replace('A) Proceed with the full split (recommended)','A) Approve the first proposal (recommended)');}, - 'selected option lacks its own comparison':(q:any)=>{q.question=q.question.replace('✅ You see and rule on all 11 proposals; none are cut by me before you weigh in','');}, - 'quoted mapping is not evidence':(q:any)=>{q.options[0].description='"'+q.options[0].description+'"';}, - 'historical inventory':(q:any)=>{q.question=q.question.replace('The 10x analysis produced','Previously the 10x analysis produced');}, - 'conditional inventory':(q:any)=>{q.question=q.question.replace('The 10x analysis produced','If the 10x analysis produced');}, - 'inventory asserts approved status':(q:any)=>{q.question=q.question.replace('L3 auto-persist last filters','L3 auto-persist last filters (already approved)');}, - 'inventory conceals an actor grant':(q:any)=>{q.question=q.question.replace('L3 auto-persist last filters','L3 auto-persist last filters; we approve all eleven now');}, - 'inventory caption imperatively approves':(q:any)=>{q.question=q.question.replace('E1 shared visibility','E1 approve all proposals');}, - 'inventory caption declares approved':(q:any)=>{q.question=q.question.replace('E1 shared visibility','E1 approved shared views');}, - 'inventory caption defers other items':(q:any)=>{q.question=q.question.replace('E1 shared visibility','E1 defer others');}, - 'inventory caption hides imperative after a noun':(q:any)=>{q.question=q.question.replace('E1 shared visibility','E1 shared visibility and approve E2');}, - 'selected immediate scope approval':(q:any)=>{q.options[0].description+=' Approve E1 now.';}, - 'selected implicit approval':(q:any)=>{q.options[0].description+=' All proposals are included.';}, - 'selected omission':(q:any)=>{q.options[0].description+=' Except E4.';}, - 'selected grouping':(q:any)=>{q.options[0].description+=' Batch E1 and E2 together.';}, - 'unconditional effect in an unselected option':(q:any)=>{q.options[1].description+=' Regardless of choice, include E1 now.';}, - 'grant concealed in task title':(q:any)=>{q.question=q.question.replace('Add saved project views','Approve all proposals now');}, - 'extra decision':(q:any)=>{q.question=q.question.replace('ELI10:','ELI10: Should we remove access checks?');}, - 'duplicate preserving option':(q:any)=>{q.options[1]=structuredClone(q.options[0]);}, - }; - for(const[name,change]of Object.entries(negative))test(name,()=>{const e=state();change(e.pacing.questions[0]);expect(choose(e)?.index).not.toBe(1);}); - test('descriptive inventory nouns remain valid and substantive scope cards remain substantive',()=>{ - const e=state();e.pacing.questions[0]!.question=e.pacing.questions[0]!.question.replace('E1 shared visibility','E1 delete history views');expect(choose(e)?.index).toBe(1); - const pending=structuredClone(f.pending),transcript={status:'ready' as const,calls:[structuredClone(f.mode),pending],assistantMessages:[]}; - expect(ceoExpansionPacingChoice(screen(pending),transcript as any,f.selectedAt)).toBeNull(); - pending.questions[0]!.question=pending.questions[0]!.question.replace(/^D3\.1[^\n]+/,'D3.1 — Should we split the shared-view proposal into separate schemas?'); - expect(ceoExpansionPacingChoice(screen(pending),transcript as any,f.selectedAt)).toBeNull(); - }); - test('mode ownership, matching pane and actual ACK remain mandatory',()=>{ - const e=state();e.pacing.sessionId='foreign';expect(choose(e)).toBeNull(); - const noMode=state();noMode.mode.answered=false;expect(choose(noMode)).toBeNull(); - const ack=state(),pick=choose(ack)!;expect(pick?.index).toBe(1); - expect(ceoExpansionPacingReady('next',ack.transcript as any,pick,[])).toBe(false); - }); -}); - - -describe('same-proposal discussion control makes no scope decision',()=>{ - const f=nativePacing77.countedNativeB955; - function state(){ - const mode=structuredClone(f.mode),proposal=structuredClone(f.pending) as NativePlanQuestionCall; - // The actual proposal stayed pending. This derived ACK exercises only the - // downstream predicate; it cannot convert the original paid timeout to PASS. - proposal.answered=true;proposal.unansweredQuestionIndices=[]; - proposal.answers={[proposal.questions[0]!.question]:proposal.questions[0]!.options[0]!.label}; - proposal.answeredAt='2026-09-15T20:44:00.000Z'; - const calls=[mode,proposal]; - const events=f.events.filter(e=>calls.some(c=>c.toolUseId===e.toolUseId)).map(e=>e.kind==='use'?{...e,input:{questions:calls.find(c=>c.toolUseId===e.toolUseId)!.questions}}:{...e}) as NativePublicToolEvent[]; - events.push({kind:'result',sessionId:proposal.sessionId,toolUseId:proposal.toolUseId,timestamp:proposal.answeredAt,isError:false}); - return{proposal,transcript:{status:'ready' as const,calls,assistantMessages:[]},events}; - } - const matches=(e=state())=>hasNativePostAnswerCeoPosture(e.transcript,'SCOPE EXPANSION',pattern,f.selectedAt,e.events); - test('actual stop-and-discuss current E1 content remains nonoperative under a synthetic Include ACK',()=>{expect(f.pending.answered).toBe(false);expect(matches()).toBe(true);}); - test.each(['Pause the review. Discuss E1 before proceeding.','Discuss E1 before continuing; stop the chain.'])('equivalent two-clause procedural control: %s',description=>{ - const e=state();e.proposal.questions[0]!.options[3]!.description=description;expect(matches(e)).toBe(true); - }); - test.each(['Stop the chain; discuss E2 before continuing.','Stop the chain; approve E1 before continuing.','Stop the chain; discuss E1 before continuing. Add E2.', - 'Discuss E1 before continuing.','Stop the chain.','"Stop the chain; discuss E1 before continuing."','Previously stop the chain; discuss E1 before continuing.', - 'If needed, stop the chain; discuss E1 before continuing.','Stop the chain; discuss E1 before implementing it.'])('foreign, incomplete or operative control stays negative: %s',description=>{ - const e=state();e.proposal.questions[0]!.options[3]!.description=description;expect(matches(e)).toBe(false); - }); - test('pending, selected Hold, duplicate and foreign ACKs still supply no posture',()=>{ - for(const change of [ - (e:ReturnType)=>{e.proposal.answered=false;}, - (e:ReturnType)=>{const q=e.proposal.questions[0]!;e.proposal.answers={[q.question]:q.options[3]!.label};}, - (e:ReturnType)=>{e.events.push({...e.events.at(-1)!});}, - (e:ReturnType)=>{e.events.at(-1)!.sessionId='foreign';}, - ]){const e=state();change(e);expect(matches(e)).toBe(false);} - }); -}); - - -test.each(['acknowledged pacing','missing pacing ACK'])('actual paid posture loop preserves the substantive allowance: %s',async scenario=>{ - const f=nativePacing77.countedNativeB955; - const source=fs.readFileSync(path.join(import.meta.dir,'skill-e2e-plan-ceo-mode-routing.test.ts'),'utf8'); - const planDeclaration=source.match(/^const PLAN = \[[\s\S]*?^\]\.join\('\\n'\);/m)?.[0]; - expect(planDeclaration).toBeDefined(); - const plan=new Function(`${planDeclaration}; return PLAN;`)(); - const start=source.indexOf(' const budgetMs = 240_000;'),end=source.indexOf(" outcome = 'posture_confirmed';",start); - expect(start).toBeGreaterThan(0);expect(end).toBeGreaterThan(start); - const loop=source.slice(start,end+" outcome = 'posture_confirmed';".length); - const keys=['Bun','Date','c','session','sincePick','selectionStartedAt','question','fixture','capture','readPlanCountTranscript', - 'readPendingQuestion','hasNativePostAnswerCeoPosture','ceoModeSubmissionInput','ceoExpansionPacingReady','ceoExpansionPacingChoice', - 'nextCeoPostureContinuation','capturePlanCountQuestion','planCountQuestionInput','selectPtyNumberedOption','isPlanReadyVisible','isNumberedOptionListVisible', - 'EXPANSION_PACING_CALLS','modeIndex','artifacts','visibleAtMode','postureSource']; - const compiled=new Bun.Transpiler({loader:'ts'}).transformSync(`async function run(b){const {${keys.join(',')}}=b;let outcome;${loop};return {outcome,continuedQuestion,pacingCalls};}`); - const run=new Function(compiled+';return run;')(); - const pending=structuredClone(f.pacing);pending.answered=false;delete pending.answers;delete pending.answeredAt;delete pending.unansweredQuestionIndices; - const proposal=structuredClone(f.pending) as NativePlanQuestionCall; - let stage=0,clock=f.selectedAt; - const sends:string[]=[]; - const snapshots:string[]=[]; - const view=()=>stage===0?f.viewport:f.nextViewport; - const session={hermeticConfigDir:'fixture-native',pendingQuestionFile:'fixture-pending',exited:()=>false,exitCode:()=>null, - currentScreen:async()=>view(),visibleSince:()=>view(),visibleText:()=>view(),send:(value:string)=>{ - sends.push(value);stage++; - if(stage===2){proposal.answered=true;proposal.answers={[proposal.questions[0]!.question]:proposal.questions[0]!.options[0]!.label}; - proposal.answeredAt='2026-09-15T20:44:00.000Z';proposal.unansweredQuestionIndices=[];} - }}; - const readPlanCountTranscript=(_config:string,_cwd:string,emit:(e:NativePublicToolEvent)=>void)=>{ - const pacing=stage===0||scenario==='missing pacing ACK'?pending:f.pacing; - const calls=stage===0?[f.mode,pacing]:[f.mode,pacing,proposal]; - const events=f.events.filter(e=>calls.some(c=>c.toolUseId===e.toolUseId)&&!(e.kind==='result'&&e.toolUseId===f.pacing.toolUseId&&!pacing.answered)) - .map(e=>e.kind==='use'?{...e,input:{questions:calls.find(c=>c.toolUseId===e.toolUseId)!.questions}}:{...e}) as NativePublicToolEvent[]; - if(proposal.answered)events.push({kind:'result',sessionId:proposal.sessionId,toolUseId:proposal.toolUseId,timestamp:proposal.answeredAt!,isError:false}); - events.forEach(emit);return{status:'ready',calls,assistantMessages:[]}; - }; - const bindings={Bun:{sleep:async(ms:number)=>{clock+=ms;}},Date:{now:()=>clock},c:{mode:'SCOPE EXPANSION',postureRe:pattern},session,sincePick:0, - selectionStartedAt:f.selectedAt,question:{nativeCall:f.mode},fixture:{cwd:'fixture-root'},capture:(state:string)=>snapshots.push(state),readPlanCountTranscript, - readPendingQuestion:()=>undefined,hasNativePostAnswerCeoPosture,ceoModeSubmissionInput,ceoExpansionPacingReady,ceoExpansionPacingChoice,nextCeoPostureContinuation, - capturePlanCountQuestion,planCountQuestionInput,selectPtyNumberedOption:async(s:any,index:number)=>s.send(String(index)),isPlanReadyVisible,isNumberedOptionListVisible, - EXPANSION_PACING_CALLS:1,modeIndex:2,artifacts:{},visibleAtMode:'captured mode menu', - postureSource:{path:path.join('fixture-root','PLAN.md'),content:plan}}; - if(scenario==='missing pacing ACK')await expect(run(bindings)).rejects.toThrow('no posture match'); - else expect(await run(bindings)).toEqual({outcome:'posture_confirmed',continuedQuestion:true,pacingCalls:1}); - expect(sends).toEqual(scenario==='missing pacing ACK'?['1']:['1','1']); - expect(snapshots.length).toBeGreaterThan(0); - expect(f.pending.answered).toBe(false); // Final synthetic ACK is never paid evidence. -}); diff --git a/test/ceo-mode-option.test.ts b/test/ceo-mode-option.test.ts index 56b97048d..69ba0b2cb 100644 --- a/test/ceo-mode-option.test.ts +++ b/test/ceo-mode-option.test.ts @@ -7,6 +7,30 @@ import * as fs from 'node:fs'; import * as os from 'node:os'; import * as path from 'node:path'; import { pathToFileURL } from 'node:url'; +import captured_ceo_hold_commitment_ar from './fixtures/ceo-hold-commitment-ar.json'; +import captured_ceo_hold_posture_ag from './fixtures/ceo-hold-posture-ag.json'; +import retainedPreservationCaptures_ceo_hold_posture_ag from './fixtures/ceo-hold-preservation-f359.json'; +import captured_ceo_mode_colon_at from './fixtures/ceo-mode-colon-at.json'; +import fs_ceo_mode_full_ad from 'node:fs'; +import os_ceo_mode_full_ad from 'node:os'; +import path_ceo_mode_full_ad from 'node:path'; +import { ceoExpansionPacingChoice } from './helpers/ceo-mode-option'; +import { ceoExpansionPacingReady } from './helpers/ceo-mode-option'; +import { ceoModeSubmissionInput } from './helpers/ceo-mode-option'; +import { capturePlanCountQuestion } from './helpers/claude-pty-runner'; +import { planCountPrerequisitePick } from './helpers/claude-pty-runner'; +import { isNumberedOptionListVisible } from './helpers/claude-pty-runner'; +import { isPlanReadyVisible } from './helpers/claude-pty-runner'; +import { readPlanCountTranscript } from './helpers/plan-count-transcript'; +import type { NativePublicToolEvent } from './helpers/plan-count-transcript'; +import type { NativePlanQuestionCall } from './helpers/plan-count-transcript'; +import captured_ceo_mode_full_ad from './fixtures/ceo-mode-full-ad.json'; +import kindCapture_ceo_mode_full_ad from './fixtures/ceo-expansion-posture-kind-dacc.json'; +import pauseCapture_ceo_mode_full_ad from './fixtures/ceo-expansion-pause-6714.json'; +import completeInventory_ceo_mode_full_ad from './fixtures/ceo-expansion-complete-inventory-6f.json'; +import nativePacing77_ceo_mode_full_ad from './fixtures/ceo-expansion-pacing-77.json'; +import captured_ceo_mode_posture_ad from './fixtures/ceo-mode-posture-ad.json'; +import captured_ceo_prerequisite_ad_v2 from './fixtures/ceo-prerequisite-ad-v2.json'; describe('CEO mode option matching', () => { test('selects option 4 from the failed Claude Code 2.1.257 menu capture', () => { @@ -483,3 +507,1329 @@ const results=await Promise.all(cases.map(async item=>{ fs.rmSync(dir,{recursive:true,force:true}); } },12000); + +describe('ceo-hold-commitment-ar', () => { +const captured = captured_ceo_hold_commitment_ar; +const posture = /\b(rigor|bulletproof|hold\s*scope|maximum\s+rigor)\b/i; +const original = captured.transcript.assistantMessages[0]!.text; +const replay = () => structuredClone(captured.transcript) as PlanCountTranscript; +const matches = (transcript = replay()) => hasNativePostAnswerCeoPosture( + transcript, 'HOLD SCOPE', posture, captured.selectionStartedAt, +); +const withText = (text: string) => { const t = replay(); t.assistantMessages[0]!.text = text; return matches(t); }; + +test('the actual failed attempt adopted HOLD through scope, hardening and exclusion', () => { + expect(captured.provenance.actualState).toBe('failed'); + expect(nativeCeoModeAnswer(replay(), 'HOLD SCOPE', captured.selectionStartedAt)?.toolUseId) + .toBe('toolu_01E1HnYjRCz79826bo7nNnoK'); + expect(posture.test(original)).toBe(false); + expect(matches()).toBe(true); + // This is prospective posture recognition, not evidence of completed work. + for (const prefix of ["I'm keeping", 'I am keeping', 'I will keep', "We'll keep", 'We will keep', 'We are keeping']) { + expect(withText(original.replace("I'll keep", prefix)), prefix).toBe(true); + } + expect(withText(original.replace("I'll", 'I’ll').replace("PLAN.md's", 'PLAN.md’s'))).toBe(true); +}); + +test('explicitly future, conditional and quoted statements are not adopted current posture', () => { + for (const text of [ + original.replace("I'll keep", 'I will later keep'), + original.replace("I'll keep", 'I will eventually keep'), + original.replace("I'll keep", 'I would keep'), + original.replace("I'll keep", 'I may keep'), + original.replace("I'll keep", "I'll not keep"), + original.replace('scope fixed', 'scope tomorrow fixed'), + original.replace('production visibility', 'production visibility next week'), + original.replace('production visibility', 'production visibility tomorrow'), + ...['after approval', 'once approved', 'when approved', 'after launch', 'pending approval', 'subject to approval'].map(when => + original.replace('production visibility', 'production visibility ' + when)), + 'Later, ' + original, 'If you approve, ' + original, + 'Hypothetical scenario. ' + original, 'Example only: ' + original, + '"' + original + '"', '> ' + original, + '```text\n' + original + '\n```', '~~~text\n' + original + '\n~~~', + 'Read(file)\n' + original, 'The user said: ' + original, + ]) expect(withText(text), text).toBe(false); +}); + +test('all three obligations remain concrete and bound to the selected plan', () => { + for (const [from, to] of [ + ['PLAN.md', 'OTHER.md'], ['PLAN.md', 'archive/PLAN.md'], + ["PLAN.md's four bullets plus the approved schema", 'the future expanded plan'], + ['plus the approved schema', 'plus a new unapproved schema'], + [', pressure-testing every stated behavior for failure modes, errors, tests, and production visibility', ''], + ['errors, tests, and production visibility', 'word choice and formatting'], + ['while deferring anything extra rather than adding it silently', 'while adding anything extra'], + ['while deferring', 'while not deferring'], ['pressure-testing', 'not pressure-testing'], + ]) expect(withText(original.replace(from!, to!)), from).toBe(false); + for (const contextChange of [ + (text: string) => text.replace('PLAN.md', 'PLAN.md and OTHER.md'), + (text: string) => text.replace('schema) approved', 'schema) not approved'), + (text: string) => text.replace('schema) approved', 'schema) discussed'), + ...['approved if the user agrees', 'approved once migration finishes', 'approved pending migration', 'approved subject to migration'].map(status => + (text: string) => text.replace('schema) approved', 'schema) ' + status)), + ]) { + const t = replay(); const q = t.calls[0]!.questions[0]!; const prior = q.question; + q.question = contextChange(q.question); t.calls[0]!.answers = { [q.question]: t.calls[0]!.answers![prior]! }; + expect(matches(t)).toBe(false); + } +}); + +test('current corrections withdraw a commitment; quoted corrections do not', () => { + for (const correction of [ + 'Correction: I will expand scope to include defaults.', + 'Correction: I will not keep scope fixed to these requirements.', + 'Correction: I am no longer keeping scope to those requirements.', + 'The formerly excluded additions are in scope.', + ]) { + expect(withText(original + '\n\n' + correction), correction).toBe(false); + for (const quote of ['> ' + correction, '```text\n' + correction + '\n```', '~~~text\n' + correction + '\n~~~', 'A quotation: "' + correction + '"']) { + expect(withText(original + '\n\n' + quote), quote).toBe(true); + } + } +}); + +test('native selection, session and timestamp evidence remain required', () => { + for (const change of [ + (t: PlanCountTranscript) => { t.status = 'missing'; }, + (t: PlanCountTranscript) => { t.calls[0]!.answered = false; }, + (t: PlanCountTranscript) => { t.calls[0]!.failed = true; }, + (t: PlanCountTranscript) => { t.calls[0]!.answeredAt = new Date(captured.selectionStartedAt - 1).toISOString(); }, + (t: PlanCountTranscript) => { t.calls[0]!.answers![t.calls[0]!.questions[0]!.question] = 'Scope expansion'; }, + (t: PlanCountTranscript) => { t.calls[0]!.answers![t.calls[0]!.questions[0]!.question] = 'Unknown'; }, + (t: PlanCountTranscript) => { t.assistantMessages[0]!.sessionId = 'foreign'; }, + (t: PlanCountTranscript) => { t.assistantMessages[0]!.timestamp = t.calls[0]!.answeredAt!; }, + (t: PlanCountTranscript) => { t.assistantMessages[0]!.timestamp = 'invalid'; }, + (t: PlanCountTranscript) => { t.assistantMessages[0]!.timestamp = new Date(Date.now() + 60_000).toISOString(); }, + (t: PlanCountTranscript) => { t.assistantMessages = []; }, + ]) { const t = replay(); change(t); expect(matches(t)).toBe(false); } +}); +}); + +describe('ceo-hold-posture-ag', () => { +const captured = captured_ceo_hold_posture_ag; +const retainedPreservationCaptures = retainedPreservationCaptures_ceo_hold_posture_ag; +const posture = /\b(rigor|bulletproof|hold\s*scope|maximum\s+rigor)\b/i; +const original = captured.transcript.assistantMessages[0]!.text; +const replay = () => structuredClone(captured.transcript) as PlanCountTranscript; +const matches = (transcript = replay()) => hasNativePostAnswerCeoPosture( + transcript, 'HOLD SCOPE', posture, captured.selectionStartedAt, +); + +test('the captured selected HOLD scope lock and hardening establish posture without a keyword', () => { + const transcript = replay(); + expect(captured.provenance.actualState).toBe('failed'); + expect(nativeCeoModeAnswer(transcript, 'HOLD SCOPE', captured.selectionStartedAt)?.toolUseId) + .toBe('toolu_011bt3yabPDSEsPNm97EhqV4'); + expect(posture.test(original)).toBe(false); + expect(matches(transcript)).toBe(true); +}); + +test('ordinary current scope declarations preserve the same three obligations', () => { + for (const text of [ + original.replace("I'm locking", 'I will lock'), + original.replace("I'm locking", "I'll lock"), + original.replace("I'm locking", 'We are keeping').replace('the four PLAN.md bullets from approach B', 'the agreed plan') + .replace('flagging anything beyond', 'treating everything outside').replace('hunting', 'checking'), + original.replace("I'm locking", 'I am holding').replace('four PLAN.md bullets from approach B', 'PLAN.md requirements') + .replace('flagging', 'marking').replace('hunting', 'looking'), + original.replace("I'm", 'I’m').replace('PLAN.md', '**PLAN.md**'), + ]) { + const transcript = replay(); transcript.assistantMessages[0]!.text = text; + expect(matches(transcript)).toBe(true); + } +}); + +test('deferred commitments, conditions and quotation cannot establish the current posture', () => { + for (const text of [ + original.replace("I'm locking", 'I would lock'), + original.replace("I'm locking", 'I will later lock'), + 'If you approve, ' + original, + 'Later, ' + original, + 'Example only: ' + original, + 'An unproven hypothesis: ' + original, + 'Example only. ' + original, + '"' + original + '"', + '> ' + original, + '```text\n' + original + '\n```', + '~~~~\n' + original + '\n~~~~', + 'Read(file)\n' + original, + 'The user said: ' + original, + original.replace('and hunting', 'and not hunting'), + ]) { + const transcript = replay(); transcript.assistantMessages[0]!.text = text; + expect(matches(transcript), text).toBe(false); + } +}); + +test('all three obligations refer to the selected current scope', () => { + for (const text of [ + original.replace('PLAN.md', 'OTHER.md'), + original.replace('PLAN.md', 'archive/PLAN.md'), + original.replace('the four PLAN.md bullets from approach B', 'the future expanded plan'), + original.replace('the four PLAN.md bullets from approach B', 'the two imagined requirements'), + original.replace('out of scope', 'in scope'), + original.replace('as out of scope', 'as not out of scope'), + original.replace('flagging anything beyond that (defaults, sharing, deep links) as out of scope, and ', ''), + original.replace(/, and hunting[^.]+\./, '.'), + original.replace('constraints, error handling, UI edge cases, access-rule leaks', 'word choice and formatting'), + original + ' I am expanding scope to include a new feature.', + original + ' I am adding extra features to scope.', + ]) { + const transcript = replay(); transcript.assistantMessages[0]!.text = text; + expect(matches(transcript), text).toBe(false); + } + const ambiguous = replay(); + const question = ambiguous.calls[0]!.questions[0]!; + const oldQuestion = question.question; + question.question = question.question.replace('reviewing PLAN.md', 'reviewing PLAN.md and OTHER.md'); + ambiguous.calls[0]!.answers = { [question.question]: ambiguous.calls[0]!.answers![oldQuestion]! }; + expect(matches(ambiguous)).toBe(false); +}); + +test('explicit later corrections withdraw scope locking, while quoted examples do not', () => { + const corrections = [ + 'Correction: the previously excluded defaults, sharing, and deep links are now in scope.', + 'Correction: I am no longer locking scope to those requirements.', + 'I am not keeping scope to those requirements.', + 'The formerly excluded additions are in scope.', + ]; + for (const correction of corrections) { + const transcript = replay(); + transcript.assistantMessages[0]!.text = original + '\n\n' + correction; + expect(matches(transcript), correction).toBe(false); + for (const quote of ['> ' + correction, '```text\n' + correction + '\n```', + '~~~text\n' + correction + '\n~~~', 'An example of withdrawn wording is: "' + correction + '"']) { + transcript.assistantMessages[0]!.text = original + '\n\n' + quote; + expect(matches(transcript), quote).toBe(true); + } + } +}); + +test('only a real selected HOLD answer followed by its own public statement supplies evidence', () => { + for (const change of [ + (t: PlanCountTranscript) => { t.status = 'missing'; }, + (t: PlanCountTranscript) => { t.calls[0]!.answered = false; }, + (t: PlanCountTranscript) => { t.calls[0]!.failed = true; }, + (t: PlanCountTranscript) => { t.calls[0]!.answeredAt = new Date(captured.selectionStartedAt - 1).toISOString(); }, + (t: PlanCountTranscript) => { t.calls[0]!.answers![t.calls[0]!.questions[0]!.question] = 'Scope Expansion'; }, + (t: PlanCountTranscript) => { t.calls[0]!.answers![t.calls[0]!.questions[0]!.question] = 'Unknown'; }, + (t: PlanCountTranscript) => { t.assistantMessages[0]!.sessionId = 'foreign'; }, + (t: PlanCountTranscript) => { t.assistantMessages[0]!.timestamp = t.calls[0]!.answeredAt!; }, + (t: PlanCountTranscript) => { t.assistantMessages[0]!.timestamp = 'invalid'; }, + (t: PlanCountTranscript) => { t.assistantMessages[0]!.timestamp = new Date(Date.now() + 60_000).toISOString(); }, + (t: PlanCountTranscript) => { t.assistantMessages = []; }, + ]) { + const transcript = replay(); change(transcript); expect(matches(transcript)).toBe(false); + } + const expansion = replay(); + expansion.calls[0]!.answers![expansion.calls[0]!.questions[0]!.question] = 'Scope Expansion'; + expect(hasNativePostAnswerCeoPosture(expansion, 'SCOPE EXPANSION', posture, captured.selectionStartedAt)).toBe(false); +}); +// Exact public AY parent narration after the answered HOLD SCOPE mode AUQ. +// Its native ownership controls use the existing PLAN.md / approved-approach-B fixture. +const ambiguityNarration = "I'm holding strictly to the plan's approved scope (Approach B, private-only views) and flagging any ambiguities the sketch leaves undecided as targeted questions rather than expanding scope. First up: what happens when a saved view's filters reference something that's been deleted.\n\n"; +const ambiguityReplay = () => { + const transcript = replay(); + transcript.assistantMessages[0]!.text = ambiguityNarration; + return transcript; +}; +const ambiguityMatches = (text = ambiguityNarration) => { + const transcript = ambiguityReplay(); transcript.assistantMessages[0]!.text = text; + return matches(transcript); +}; + +test('approved scope plus targeted ambiguity questions applies HOLD without naming the mode', () => { + expect(posture.test(ambiguityNarration)).toBe(false); + expect(ambiguityMatches()).toBe(true); + for (const text of [ + ambiguityNarration.replace("I'm holding", 'We are keeping'), + ambiguityNarration.replace("I'm holding", 'I will hold'), + ambiguityNarration.replace('the sketch leaves undecided', 'in the plan').replace('flagging', 'surfacing'), + ambiguityNarration.replace("plan's", "PLAN.md's"), + ambiguityNarration.replace("I'm", 'I’m').replace("plan's", 'plan’s'), + ]) expect(ambiguityMatches(text), text).toBe(true); +}); + +test('ambiguity wording must adopt every obligation without quoting, negating or deferring it', () => { + for (const text of [ + '> ' + ambiguityNarration, '"' + ambiguityNarration.trim() + '"', + '```text\n' + ambiguityNarration + '```', '~~~text\n' + ambiguityNarration + '~~~', + 'Example only: ' + ambiguityNarration, 'The user said: ' + ambiguityNarration, + 'Read(file)\n' + ambiguityNarration, 'If approved, ' + ambiguityNarration, + ambiguityNarration.replace("I'm holding", 'I would hold'), + ambiguityNarration.replace("I'm holding", 'I will later hold'), + ambiguityNarration.replace("I'm holding", "I'm not holding"), + ambiguityNarration.replace('and flagging', 'and not flagging'), + ambiguityNarration.replace('approved scope', 'proposed scope'), + ambiguityNarration.replace("plan's", "OTHER.md's"), + ambiguityNarration.replace('private-only views', 'OTHER.md views'), + ambiguityNarration.replace('Approach B', 'Approach C'), + ambiguityNarration.replace('as targeted questions rather than expanding scope', 'as optional improvements'), + ambiguityNarration.replace('rather than expanding scope', 'while expanding scope'), + ambiguityNarration.replace('ambiguities the sketch leaves undecided', 'word choice and formatting'), + ]) expect(ambiguityMatches(text), text).toBe(false); + for (const correction of [ + 'I am expanding scope to include sharing.', + 'Correction: I will add defaults to scope.', + 'The previously excluded sharing feature is now in scope.', + 'Correction: I am no longer holding scope to this plan.', + "Correction: I am not holding strictly to the plan's approved scope.", + 'Correction: I am no longer flagging ambiguities as targeted questions.', + 'Correction: this posture is withdrawn.', + 'This posture is no longer current.', + ]) { + expect(ambiguityMatches(ambiguityNarration + correction), correction).toBe(false); + expect(ambiguityMatches(ambiguityNarration + '> ' + correction), correction).toBe(true); + } +}); + +test('ambiguity posture stays bound to the approved plan and its actual native answer', () => { + for (const change of [ + (t: PlanCountTranscript) => { t.calls[0]!.answered = false; }, + (t: PlanCountTranscript) => { t.calls[0]!.failed = true; }, + (t: PlanCountTranscript) => { t.calls[0]!.answers![t.calls[0]!.questions[0]!.question] = 'Scope Expansion'; }, + (t: PlanCountTranscript) => { t.assistantMessages[0]!.sessionId = 'foreign'; }, + (t: PlanCountTranscript) => { t.assistantMessages[0]!.timestamp = t.calls[0]!.answeredAt!; }, + (t: PlanCountTranscript) => { t.assistantMessages[0]!.timestamp = 'invalid'; }, + (t: PlanCountTranscript) => { t.assistantMessages[0]!.timestamp = new Date(Date.now() + 60_000).toISOString(); }, + ]) { const transcript = ambiguityReplay(); change(transcript); expect(matches(transcript)).toBe(false); } + for (const [from, to] of [ + ['PLAN.md', 'PLAN.md and OTHER.md'], + ['approved.', 'not approved.'], + ['approved.', 'approved if accepted.'], + ['approved.', 'discussed.'], + ]) { + const transcript = ambiguityReplay(); const q = transcript.calls[0]!.questions[0]!; + const before = q.question; q.question = before.replace(from!, to!); + transcript.calls[0]!.answers = { [q.question]: transcript.calls[0]!.answers![before]! }; + expect(matches(transcript), to).toBe(false); + } +}); +{ +const captures = retainedPreservationCaptures; +const posture=/\b(rigor|bulletproof|hold\s*scope|maximum\s+rigor)\b/i; +const clone=(i=0)=>structuredClone(captures[i]) as any; +const check=(x:any)=>hasNativePostAnswerCeoPosture(x.transcript,'HOLD SCOPE',posture,x.selectionStartedAt,x.tools,x.source); +const decision=(x:any)=>x.transcript.calls.find((c:any)=>c.questions[0]?.question.match(/^D\d+ — Keep/)); +function editQuestion(x:any,change:(q:any)=>void){const c=decision(x);const before=c.questions[0].question;change(c.questions[0]);const after=c.questions[0].question;if(before!==after){c.answers[after]=c.answers[before];delete c.answers[before]};x.tools.find((t:any)=>t.kind==='use'&&t.toolUseId===c.toolUseId).input.questions=structuredClone(c.questions)} +for(let i=0;i<2;i++)test(`actual acknowledged preserve decision ${i+1}`,()=>{const x=clone(i);expect(check(x)).toBe(true)}); +const mutations:Recordvoid>={ + 'unanswered':x=>{decision(x).answered=false}, + 'failed answer':x=>{x.tools.find((t:any)=>t.kind==='result'&&t.toolUseId===decision(x).toolUseId).isError=true}, + 'unmatched native request':x=>{x.tools.find((t:any)=>t.kind==='use'&&t.toolUseId===decision(x).toolUseId).input.questions=[]}, + 'foreign decision session':x=>{decision(x).sessionId='foreign'}, + 'foreign source path':x=>{x.source.path='/foreign/PLAN.md'}, + 'altered source bytes':x=>{x.source.content=x.source.content.replace('update,','share,')}, + 'different named source':x=>{editQuestion(x,q=>q.question=q.question.replace('PLAN.md','OTHER.md'))}, + 'unrelated choice':x=>{editQuestion(x,q=>{q.question=q.question.replaceAll('update','sharing');q.options=q.options.map((o:any)=>({...o,label:o.label.replaceAll('update','sharing')}))});const c=decision(x);c.answers[c.questions[0].question]=c.questions[0].options[0].label}, + 'expanding description':x=>{editQuestion(x,q=>q.options[0].description+=' Also add shared team views outside the plan.')}, + 'mere mode label':x=>{editQuestion(x,q=>{q.question=q.question.replace(/ELI10:[\s\S]*?Stakes if/,'ELI10: Keep it.\nStakes if').replace(/Stakes if[\s\S]*?Recommendation:/,'Stakes if we pick wrong: None.\nRecommendation:');q.options.forEach((o:any)=>o.description='Fine.')})}, + 'historical decision':x=>{editQuestion(x,q=>q.question='Historical example: '+q.question)}, + 'quoted decision':x=>{editQuestion(x,q=>q.question=q.question.split('\n').map((l:string)=>'> '+l).join('\n'))}, + 'withdrawn decision':x=>{editQuestion(x,q=>q.question=q.question.replace('HOLD SCOPE review','withdrawn HOLD SCOPE review'))}, + 'later withdrawal':x=>{x.transcript.assistantMessages.push({sessionId:decision(x).sessionId,timestamp:new Date().toISOString(),text:'I withdraw this decision.'})}, + 'later scope expansion':x=>{x.transcript.assistantMessages.push({sessionId:decision(x).sessionId,timestamp:new Date().toISOString(),text:'I expand the scope.'})}, + 'missing source ACK':x=>{x.tools=x.tools.filter((t:any)=>!(t.kind==='result'&&x.tools.some((u:any)=>u.kind==='use'&&u.toolUseId===t.toolUseId&&u.name==='Read'&&u.input?.file_path===x.source.path)))}, + 'wrong actual choice':x=>{const c=decision(x);c.answers[c.questions[0].question]=c.questions[0].options[1].label}, +}; +for(const [name,mutate] of Object.entries(mutations))test(name,()=>{const x=clone();mutate(x);expect(check(x)).toBe(false)}); +test('later quoted withdrawal is not current withdrawal',()=>{const x=clone();x.transcript.assistantMessages.push({sessionId:decision(x).sessionId,timestamp:new Date().toISOString(),text:'Example: "I withdraw this decision."'});expect(check(x)).toBe(true)}); +test('new proof path is unavailable without explicit fixture source binding',()=>{const x=clone();expect(hasNativePostAnswerCeoPosture(x.transcript,'HOLD SCOPE',posture,x.selectionStartedAt,x.tools)).toBe(false)}); + +test('retry source cat requires the actual owned project',()=>{const x=clone(1);x.tools.find((t:any)=>t.kind==='use'&&t.input?.command?.includes('cat PLAN.md')).input.command=x.tools.find((t:any)=>t.kind==='use'&&t.input?.command?.includes('cat PLAN.md')).input.command.replace(x.source.path.replace('/PLAN.md',''),'/foreign');expect(check(x)).toBe(false)}); +test('retry source read ACK cannot be missing',()=>{const x=clone(1);const use=x.tools.find((t:any)=>t.kind==='use'&&t.input?.command?.includes('cat PLAN.md'));x.tools=x.tools.filter((t:any)=>!(t.kind==='result'&&t.toolUseId===use.toolUseId));expect(check(x)).toBe(false)}); + +} +}); + +describe('ceo-mode-colon-at', () => { +const captured = captured_ceo_mode_colon_at; +function transcript(): PlanCountTranscript { + return { status: 'ready', calls: [structuredClone(captured)], assistantMessages: [] }; +} + +describe('CEO colon-prefixed native mode choices', () => { + test('the exact public menu resolves each named mode by display position', () => { + const options = captured.questions[0]!.options.map((option, i) => ({ index: i + 1, label: option.label })); + expect(findCeoModeOption(options, 'SELECTIVE EXPANSION')).toBe(1); + expect(findCeoModeOption(options, 'SCOPE EXPANSION')).toBe(2); + expect(findCeoModeOption(options, 'HOLD SCOPE')).toBe(3); + expect(findCeoModeOption(options, 'SCOPE REDUCTION')).toBe(4); + }); + + test('navigation selects expansion in either display order without changing native input', () => { + for (const reverse of [false, true]) { + const call = transcript().calls[0]!; + call.answered = false; + delete call.answers; + delete call.unansweredQuestionIndices; + const question = call.questions[0]!; + if (reverse) question.options.reverse(); + const original = structuredClone(call); + const visible = `☐ ${question.header}\n${question.question}\n` + question.options.map((option, i) => + `${i ? ' ' : '❯'} ${i + 1}. ${option.label}`).join('\n') + + '\nEnter to select · ↑/↓ to navigate · Esc to cancel'; + const action = nextCeoModeNavigation(visible, 'SCOPE EXPANSION', new Set(), call); + expect(action.kind).toBe('mode'); + expect(action.kind === 'mode' && action.index).toBe(reverse ? 3 : 2); + expect(call).toEqual(original); + } + }); + + test('the recorded wrong selection remains selective expansion, never expansion coverage', () => { + const actual = transcript(); + expect(nativeCeoModeAnswer(actual, 'SELECTIVE EXPANSION', 0)?.toolUseId) + .toBe('toolu_01XY3qPeSuJZa3H2uCfatJ8b'); + expect(nativeCeoModeAnswer(actual, 'SCOPE EXPANSION', 0)).toBeNull(); + expect(actual.calls[0]).toEqual(captured); + }); + + test('pending, failed, stale and ambiguous native answers cannot prove selection', () => { + for (const change of [ + (value: PlanCountTranscript) => { value.calls[0]!.answered = false; }, + (value: PlanCountTranscript) => { value.calls[0]!.failed = true; }, + (value: PlanCountTranscript) => { delete value.calls[0]!.answers; }, + (value: PlanCountTranscript) => { value.calls[0]!.answeredAt = 'invalid'; }, + (value: PlanCountTranscript) => { + value.calls[0]!.questions[0]!.options.push({ label: 'E: SELECTIVE EXPANSION' }); + }, + ]) { + const value = transcript(); + change(value); + expect(nativeCeoModeAnswer(value, 'SELECTIVE EXPANSION', 0)).toBeNull(); + } + expect(nativeCeoModeAnswer(transcript(), 'SELECTIVE EXPANSION', Date.parse(captured.answeredAt) + 1)).toBeNull(); + const laterAmbiguous = transcript(); + const later = structuredClone(laterAmbiguous.calls[0]!); + later.toolUseId = 'later-ambiguous-mode'; + later.answeredAt = new Date(Date.parse(captured.answeredAt) + 1000).toISOString(); + later.questions[0]!.options.push({ label: 'E: SELECTIVE EXPANSION' }); + laterAmbiguous.calls.push(later); + expect(nativeCeoModeAnswer(laterAmbiguous, 'SELECTIVE EXPANSION', 0)).toBeNull(); + }); + + test('action titles, lookalikes and preview descriptions do not become modes', () => { + for (const label of [ + 'A: Use HOLD SCOPE for the next review', + 'B: Explain SCOPE EXPANSION', + 'AA: HOLD SCOPE', + '1: HOLD SCOPE', + 'A:: HOLD SCOPE', + 'A: HOLD SCOPES', + 'A: Fix contrast │ HOLD SCOPE', + 'A: Fix contrast ┌ SCOPE EXPANSION', + 'A: "HOLD SCOPE"', + 'Prior: HOLD SCOPE', + ]) expect(findCeoModeOption([{ index: 1, label }], 'HOLD SCOPE')).toBeNull(); + expect(() => findCeoModeOption([ + { index: 1, label: 'A: SELECTIVE EXPANSION │ SCOPE EXPANSION' }, + { index: 2, label: 'B: HOLD SCOPE' }, + ], 'SCOPE EXPANSION')).toThrow('not in option labels'); + }); + + test('duplicate and missing mode titles fail before selection; legacy prefixes still work', () => { + expect(() => findCeoModeOption([ + { index: 1, label: 'A: HOLD SCOPE' }, + { index: 2, label: 'HOLD SCOPE (recommended)' }, + ], 'HOLD SCOPE')).toThrow('duplicate'); + expect(() => findCeoModeOption([{ index: 1, label: 'A: SCOPE REDUCTION' }], 'HOLD SCOPE')) + .toThrow('not in option labels'); + for (const label of ['A) HOLD SCOPE', 'A. HOLD SCOPE', 'a: hold scope', 'A: HOLD SCOPE']) { + expect(findCeoModeOption([{ index: 3, label }], 'HOLD SCOPE')).toBe(3); + } + }); +}); +}); + +describe('ceo-mode-full-ad', () => { +const fs = fs_ceo_mode_full_ad; +const os = os_ceo_mode_full_ad; +const path = path_ceo_mode_full_ad; +const captured = captured_ceo_mode_full_ad; +const kindCapture = kindCapture_ceo_mode_full_ad; +const pauseCapture = pauseCapture_ceo_mode_full_ad; +const completeInventory = completeInventory_ceo_mode_full_ad; +const nativePacing77 = nativePacing77_ceo_mode_full_ad; +const pattern=/\b(expansion|10x|delight|dream|cathedral|opt[\s-]?in)\b/i; +function replay(i:number){ + const item=captured.cases[i]!,root=fs.mkdtempSync(path.join(os.tmpdir(),'ceo-full-ad-')); + fs.mkdirSync(path.join(root,'projects','owned'),{recursive:true}); + fs.writeFileSync(path.join(root,'projects','owned',item.process.sessionId+'.jsonl'),item.records.map(r=>JSON.stringify(r)).join('\n')+'\n'); + const events:NativePublicToolEvent[]=[]; + try{return {item,transcript:readPlanCountTranscript(root,item.process.cwd,e=>events.push(e)),events};} + finally{fs.rmSync(root,{recursive:true,force:true});} +} +function pending(){const c=structuredClone(replay(0).transcript.calls[0]!);c.answered=false;delete c.answers;delete c.answeredAt;delete c.unansweredQuestionIndices;return c;} +// Full panes projected from exact native questions, not retained historical viewports. +function pane(call:NativePlanQuestionCall,index:number){const q=call.questions[index]!;return [ + call.questions.length>1?'← '+call.questions.map((v,i)=>`${i`${i?' ':'❯'} ${i+1}. ${v.label}`), + `Enter to select · ${call.questions.length>1?'Tab/Arrow keys':'↑/↓'} to navigate · Esc to cancel`].join('\n');} +function frame(c:NativePlanQuestionCall,index:number){const visible=pane(c,index);return {visible,active:capturePlanCountQuestion(visible,new Set(),0,true,c)!,routing:nativePlanCallFingerprint(c,0,true)};} +function match(e= replay(1)){return hasNativePostAnswerCeoPosture(e.transcript,'SCOPE EXPANSION',pattern,e.item.selectedAt!,e.events);} +function rebind(e:ReturnType){const d=e.transcript.calls[1]!,q=d.questions[0]!;e.events[2]!.input={questions:d.questions};d.answers={[q.question]:q.options[0]!.label};} +describe('full AD mode failures retain their actual outcomes',()=>{ + test('Proposal 1 is a completed scope decision after the actual selected mode',()=>{ + const e=replay(1);expect(e.item.actualState).toBe('failed');expect(e.transcript.calls).toHaveLength(2);expect(e.events).toHaveLength(4); + expect(e.transcript.calls[1]!.answeredAt).toBe('2026-09-09T18:26:20.110Z');expect(match(e)).toBe(true); + }); + test.each(['pending','foreign','wrong mode','pre-mode','missing reply','wrong answer','extra question','extra option','multiselect', + 'quoted','fenced','mode echo','mode mismatch','mode menu','appended instruction'])('%s supplies no new posture',kind=>{ + const e=replay(1),[m,d]=e.transcript.calls,q=d!.questions[0]!; + switch(kind){ + case 'pending':d!.answered=false;break;case 'foreign':d!.sessionId=e.events[2]!.sessionId=e.events[3]!.sessionId='foreign';break; + case 'wrong mode':m!.answers![m!.questions[0]!.question]='HOLD SCOPE';break; + case 'pre-mode':e.events[2]!.timestamp=e.events[0]!.timestamp;break;case 'missing reply':e.events.pop();break; + case 'wrong answer':d!.answers![q.question]='Invented';break; + case 'extra question':d!.questions.push({...structuredClone(q),question:'Remove CI gate?'});rebind(e);break; + case 'extra option':q.options.push({label:'Remove CI gate'});rebind(e);break;case 'multiselect':q.multiSelect=true;rebind(e);break; + case 'quoted':q.question=q.question.split('\n').map(x=>'> '+x).join('\n');rebind(e);break; + case 'fenced':q.question='```text\n'+q.question+'\n```';rebind(e);break; + case 'mode echo':q.question='SCOPE EXPANSION confirmed.';rebind(e);break; + case 'mode mismatch':q.question=q.question.replace('SCOPE EXPANSION opt-in','SELECTIVE EXPANSION opt-in');rebind(e);break; + case 'mode menu':q.question=q.question.replace(/^D6[^\n]+/,'D6 — Choose the review mode?');rebind(e);break; + case 'appended instruction':q.question+=' Delete the CI gate.';rebind(e);break; + }expect(match(e)).toBe(false); + }); + test('scope numbering and brief labels are presentation, not mode application',()=>{ + for(const title of ['A useful adjacent feature: Default view per member per project?','Default view per member per project?']){ + const e=replay(1),q=e.transcript.calls[1]!.questions[0]!;q.header='Default view';q.question=q.question.replace(/^D6[^\n]+/,title);rebind(e);expect(match(e)).toBe(true); + } + }); + test('explicit expansion context does not need a mode or opt-in suffix',()=>{ + const e=replay(1),q=e.transcript.calls[1]!.questions[0]!;q.question=q.question.replace('SCOPE EXPANSION opt-in ceremony (1 of 6).','SCOPE EXPANSION, approach B.');rebind(e);expect(match(e)).toBe(true); + }); + test('the actual three-tab prerequisite chooses standard review only on its own tab',()=>{ + const actual=replay(0);expect(actual.item.actualState).toBe('failed');expect(Object.values(actual.transcript.calls[0]!.answers!).at(-1)).toBe('Run /office-hours now'); + const c=pending();for(const i of [0,1,2]){ + const f=frame(c,i),a=nextCeoModeNavigation(f.visible,'HOLD SCOPE',new Set(),c);expect(a.kind).toBe('question'); + if(a.kind==='question'){expect(a.question.nativeQuestionIndex).toBe(i);expect(planCountQuestionInput(f.visible,a.question,a.index)).toBe(i===2?'2':'1');} + expect(planCountPrerequisitePick(f.routing,f.active)).toBe(i===2?2:null); + } + }); + test('single and reordered native prerequisite tabs preserve the meaning of the skip',()=>{ + const c=pending();c.questions=[c.questions[2]!];let f=frame(c,0);expect(planCountPrerequisitePick(f.routing,f.active)).toBe(2); + c.questions[0]!.options.reverse();f=frame(c,0);expect(planCountPrerequisitePick(f.routing,f.active)).toBe(1); + }); + test.each(['wrong tab','wrong signature','wrong body','wrong order','no metadata','completed','failed','extra action','multiselect','conditional','extra remedy','no description'])('a %s cannot borrow the prerequisite action',kind=>{ + const c=pending();if(kind==='completed')c.answered=true;if(kind==='failed')c.failed=true; + if(kind==='extra action')c.questions[2]!.options.push({label:'Accept risk'}); + if(kind==='multiselect')c.questions[2]!.multiSelect=true; + if(kind==='conditional')c.questions[2]!.options[1]!.description+=' if all tests pass.'; + if(kind==='extra remedy')c.questions[2]!.options[1]!.description+=' Remove the CI gate.'; + if(kind==='no description')c.questions[2]!.options[1]!.description=''; + const f=frame(c,2);let a=f.active; + if(kind==='wrong tab')a={...a,nativeQuestionIndex:0};if(kind==='wrong signature')a={...a,signature:'foreign:tool:question:2'}; + if(kind==='wrong body')a={...a,promptSnippet:'Choose a product direction.'};if(kind==='wrong order')a={...a,options:[...a.options].reverse()}; + if(kind==='no metadata')a={...a,nativeCall:undefined}; + expect(planCountPrerequisitePick(f.routing,a)).toBeNull(); + }); +}); + +describe('full AD HOLD retry completed sequencing rationale',()=>{ + function hold(){const e=replay(2);return {e,decision:e.transcript.calls[2]!,q:e.transcript.calls[2]!.questions[0]!};} + function matches(e:ReturnType){return hasNativePostAnswerCeoPosture(e.transcript,'HOLD SCOPE',/\b(rigor|bulletproof|hold\s*scope|maximum\s+rigor)\b/i,e.item.selectedAt!,e.events);} + function bind(e:ReturnType){const d=e.transcript.calls[2]!,q=d.questions[0]!;e.events[4]!.input={questions:d.questions};d.answers={[q.question]:q.options[0]!.label};} + test('the actual completed rationale applies HOLD to work in the previously approved approach',()=>{ + const {e,decision,q}=hold();expect(e.item.actualState).toBe('failed');expect(e.transcript.calls).toHaveLength(3); + const approach=e.transcript.calls[0]!;expect(Object.values(approach.answers!)).toEqual(['B: ViewState schema (recommended)']); + expect(approach.questions[0]!.options[0]!.description).toContain('URL params'); + expect(decision.answeredAt).toBe('2026-09-09T18:35:05.273Z');expect(q.question).toContain('not new scope either way');expect(matches(e)).toBe(true); + }); + test('three and four alternatives still express one completed review decision',()=>{ + for(const count of [3,4]){const {e,q}=hold();q.options.push({label:'Gate URL sync for the pilot'});if(count===4)q.options.push({label:'Run a limited URL sync pilot'});bind(e);expect(matches(e)).toBe(true);} + }); + test.each(['pending','foreign','before mode','missing reply','failed reply','wrong answer','metadata only','bare echo','other mode', + 'quoted rationale','fenced rationale','duplicate options','extra question','extra instruction','multiselect'])('%s is not completed HOLD rationale',kind=>{ + const {e,decision,q}=hold(); + switch(kind){ + case 'pending':decision.answered=false;break;case 'foreign':decision.sessionId=e.events[4]!.sessionId=e.events[5]!.sessionId='foreign';break; + case 'before mode':e.events[4]!.timestamp=e.events[0]!.timestamp;break;case 'missing reply':e.events.pop();break;case 'failed reply':e.events[5]!.isError=true;break; + case 'wrong answer':decision.answers![q.question]='Invented';break; + case 'metadata only':q.question=q.question.replace(/ELI10:[\s\S]*?\nStakes/,'ELI10: We will implement the URL codec.\nStakes');bind(e);break; + case 'bare echo':q.question=q.question.replace(/ELI10:[\s\S]*?\nStakes/,'ELI10: HOLD SCOPE confirmed.\nStakes');bind(e);break; + case 'other mode':q.question=q.question.replace(/HOLD SCOPE/g,'SCOPE EXPANSION');bind(e);break; + case 'quoted rationale':q.question=q.question.replace('ELI10: Approach','ELI10:\n> Approach');bind(e);break; + case 'fenced rationale':q.question=q.question.replace('ELI10: Approach','ELI10: ```Approach');bind(e);break; + case 'duplicate options':q.options[1]!.label=q.options[0]!.label;bind(e);break; + case 'extra question':decision.questions.push({...structuredClone(q),question:'Remove CI?'});bind(e);break; + case 'extra instruction':q.question+=' Disable authentication.';bind(e);break; + case 'multiselect':q.multiSelect=true;bind(e);break; + }expect(matches(e)).toBe(false); + }); +}); +describe('completed expansion disposition classes from the retained dacc public questions', () => { + // Request/answer content is captured. The envelopes and chronology below are + // synthetic: missing original JSONL timestamps must never become E2E evidence. + function current(kind: 'retry' | 'meta' | 'unanswered' = 'retry') { + const e = replay(1), decision = e.transcript.calls[1]!; + decision.questions = [structuredClone(kind === 'meta' ? kindCapture.firstMetaQuestion + : kind === 'unanswered' ? kindCapture.firstUnansweredQuestion : kindCapture.retryQuestion)]; + e.events[2]!.input = { questions: decision.questions }; + decision.answers = { [decision.questions[0]!.question]: kind === 'meta' + ? kindCapture.firstMetaAnswer : kindCapture.retryAnswer }; + if (kind === 'unanswered') { decision.answered = false; delete decision.answers; e.events.pop(); } + return e; + } + function amend(e: ReturnType, fn: (q: NativePlanQuestionCall['questions'][number]) => void) { + const d=e.transcript.calls[1]!,q=d.questions[0]!,answer=d.answers?.[q.question]; + fn(q);e.events[2]!.input={questions:d.questions};d.answers={[q.question]:answer!}; + } + test('the exact acknowledged Include content supplies posture in a synthetic ownership envelope', () => { + const e=current();expect(kindCapture.actualOutcome).toContain('Both EXPANSION attempts failed'); + expect(e.transcript.assistantMessages.every(m=>Date.parse(m.timestamp) { + const e=current();amend(e,q=>{ + if(kind==='canonical three'){ + q.options=q.options.slice(0,3).map((o,i)=>({...o,label:["A) Add to this plan's scope (recommended)",'B) Defer to TODOS.md','C) Skip'][i]!})); + } + if(kind==='reordered')q.options.reverse(); + if(kind==='curly scenario')q.question=q.question.replace('"can you share your view?"','“can you share your view?”'); + if(kind==='coverage scores')q.question=q.question.replace('Note: options differ in kind, not coverage — no completeness score.','Completeness: A=10/10, B=7/10, C=3/10'); + }); + const d=e.transcript.calls[1]!,q=d.questions[0]!; + if(kind==='canonical three')d.answers={[q.question]:q.options[0]!.label}; + if(kind==='defer')d.answers={[q.question]:q.options[1]!.label}; + if(kind==='cut')d.answers={[q.question]:q.options[2]!.label}; + expect(match(e)).toBe(true); + }); + test.each(['meta','unanswered'] as const)('the original %s does not supply completed expansion evidence', kind=>{ + expect(match(current(kind))).toBe(false); + }); + test.each(['pending','selected pause','only pause','missing core','extra action','duplicate disposition', + 'generic continuation','second question','quoted decision','fenced decision','mixed packet', + 'multiselect','missing comparison','invalid score','both comparison branches','wrong mode','missing reply'] as const)( + '%s is not a completed expansion decision', kind=>{ + const e=current();amend(e,q=>{ + if(kind==='only pause')q.options=[q.options[3]!]; + if(kind==='missing core')q.options.splice(1,1); + if(kind==='extra action')q.options[3]!.label='Remove the CI gate'; + if(kind==='duplicate disposition')q.options[3]!.label='Add to scope'; + if(kind==='generic continuation')q.question=q.question.replace(/^D3\.1[^\n]+/,'D3.1 — Continue the review?'); + if(kind==='second question')q.question=q.question.replace('\nStakes if', '\nShould we remove access checks?\nStakes if'); + if(kind==='quoted decision')q.question=q.question.split('\n').map(l=>'> '+l).join('\n'); + if(kind==='fenced decision')q.question='```text\n'+q.question+'\n```'; + if(kind==='multiselect')q.multiSelect=true; + if(kind==='missing comparison')q.question=q.question.replace('Note: options differ in kind, not coverage — no completeness score.','No comparison.'); + if(kind==='invalid score')q.question=q.question.replace('Note: options differ in kind, not coverage — no completeness score.','Completeness: A=11/10, B=7/10, C=3/10'); + if(kind==='both comparison branches')q.question=q.question.replace('\nNet:','\nCompleteness: A=10/10, B=7/10, C=3/10\nNet:'); + }); + const d=e.transcript.calls[1]!,q=d.questions[0]!; + if(kind==='pending')d.answered=false; + if(kind==='selected pause')d.answers={[q.question]:q.options[3]!.label}; + if(kind==='mixed packet'){d.questions.push({...structuredClone(q),question:'Remove access checks?'});e.events[2]!.input={questions:d.questions};} + if(kind==='wrong mode'){const m=e.transcript.calls[0]!;m.answers={[m.questions[0]!.question]:'HOLD SCOPE'};} + if(kind==='missing reply')e.events.pop(); + expect(match(e)).toBe(false); + }); +}); + + +describe('owned expansion decisions with a nondecision discussion control', () => { + function current() { + const transcript = { status: 'ready' as const, calls: structuredClone(pauseCapture.calls), assistantMessages: [] }; + const events = structuredClone(pauseCapture.events) as NativePublicToolEvent[]; + for (const event of events) if (event.kind === 'use') event.input = { questions: transcript.calls.find(c => c.toolUseId === event.toolUseId)!.questions }; + return { transcript, events }; + } + function accepted(e = current()) { return hasNativePostAnswerCeoPosture(e.transcript, 'SCOPE EXPANSION', pattern, pauseCapture.selectedAt, e.events); } + test('the captured completed Add is posture evidence; the unchosen Hold qualifier does not change its action', () => { + const e = current(); + expect(e.transcript.calls[0]!.answeredAt).toBe('2026-09-15T12:33:17.286Z'); + expect(e.transcript.calls[1]!.answeredAt).toBe('2026-09-15T12:34:22.430Z'); + expect(e.events[2]!.timestamp).toBe('2026-09-15T12:34:20.084Z'); + expect(e.transcript.calls[1]!.answers[e.transcript.calls[1]!.questions[0]!.question]).toBe('Add to scope (recommended)'); + expect(accepted(e)).toBe(true); + }); + test.each([ + ['Pause — stop the review and discuss', 'Pauses the review for clarification. No scope decision is made. Delays the remaining questions.'], + ['D) Hold: discuss first', 'Stops here so we can talk through the constraints. Nothing is approved yet. Delays this review.'], + ['Pause (wait for clarification)', 'Waits for clarification before deciding. No disposition is recorded yet.'], + ['Hold', ''], + ])('procedural label %s remains a nondecision control', (label, description) => { + const e=current(),option=e.transcript.calls[1]!.questions[0]!.options[3]!; + option.label=label;option.description=description; + expect(accepted(e)).toBe(true); + }); + test.each([ + ['Hold and add Redis', 'Pauses the review. No decision is made.'], + ['Pause (approve the proposal)', 'Waits for discussion. Nothing is decided.'], + ['Hold (roll back deployment)', 'Pauses the review. No disposition is recorded.'], + ['Continue', 'Pauses the review. No decision is made.'], + ['Hold', 'Pauses the review and adds Redis. Nothing is decided.'], + ['Pause', 'Waits for discussion. No decision is made and include Redis caching.'], + ['Hold', 'Stops the chain. No decision is made. Then deploy the new cache.'], + ['Hold', 'Pauses the review and silently approves the proposal. No decision is recorded.'], + ['Pause', 'Waits for discussion. "No decision is made."'], + ['Pause', "Waits for discussion. 'No decision is made.'"], + ['Pause', 'Waits for discussion. ‘No decision is made.’'], + ['Pause', 'Waits for discussion. “No decision is made.”'], + ['Hold', 'Stops here for discussion, then chooses the default.'], + ['Hold', 'Pauses this review. No choice is recorded. "Add Redis caching" will also happen.'], + ])('action-bearing or unproved control %s does not supply posture evidence (%s)', (label,description) => { + const e=current(),option=e.transcript.calls[1]!.questions[0]!.options[3]!; + option.label=label;option.description=description; + expect(accepted(e)).toBe(false); + }); + test('selecting the valid discussion control is still not a completed substantive disposition', () => { + const e=current(),c=e.transcript.calls[1]!,q=c.questions[0]!;c.answers={[q.question]:q.options[3]!.label}; + expect(accepted(e)).toBe(false); + }); + test('the actual capture still requires its owned successful acknowledgment', () => { + const e=current();e.events.pop();expect(accepted(e)).toBe(false); + }); +}); + + +describe('EXPANSION pacing preserves one separate substantive continuation', () => { + const retry=pauseCapture.retry; + function current() { + const mode=structuredClone(retry.mode),pacing=structuredClone(retry.pacing); + pacing.answered=false;delete (pacing as any).answers;delete (pacing as any).answeredAt;delete (pacing as any).unansweredQuestionIndices; + const transcript={status:'ready' as const,calls:[mode,pacing],assistantMessages:[]}; + return {transcript,pacing,visible:pane(pacing as NativePlanQuestionCall,0)}; + } + function choice(e=current()) {return ceoExpansionPacingChoice(e.visible,e.transcript,retry.selectedAt);} + // Canonical panes below are projected from the exact native request. The + // CLI 2.1.251 redraw stream retained these two built-ins, not a stable frame. + function withNativeControls(e=current()) { + e.visible=e.visible.replace('Enter to select','4. Type something.\n5. Chat about this\nEnter to select'); + return e; + } + test('the observed native pacing controls do not become authored choices',()=>{ + expect(choice(withNativeControls())?.index).toBe(1); + }); + test.each(['Choosing Full per-item split approves E1 immediately.', + 'Answering this question authorizes every proposed expansion.', + 'This answer commits E1 to the implementation scope.', + 'Choosing Full per-item split deploys E1 immediately.', + 'This answer ships E1 immediately.', + 'Choosing Full per-item split enables E1.', + 'This answer disables E2.', + '“Choosing Full per-item split approves E1 immediately.”'])('whole-question scope effect is not pacing: %s',effect=>{ + const e=current();e.pacing.questions[0]!.question=e.pacing.questions[0]!.question.replace('ELI10:',`ELI10: ${effect}`); + e.visible=pane(e.pacing as NativePlanQuestionCall,0);expect(choice(e)?.index).toBe(0); + }); + test.each(['unknown action','reordered controls','extra control','mismatched authored option'])('native pacing pane rejects %s',kind=>{ + const e=withNativeControls(); + if(kind==='unknown action')e.visible=e.visible.replace('Type something.','Approve all now.'); + if(kind==='reordered controls')e.visible=e.visible.replace('Type something.','Chat about this').replace('5. Chat about this','5. Type something.'); + if(kind==='extra control')e.visible=e.visible.replace('Enter to select','6. More actions\nEnter to select'); + if(kind==='mismatched authored option')e.visible=e.visible.replace('Full per-item split','Approve all proposals'); + expect(choice(e)?.index).toBe(0); + }); + test('the captured full-per-item answer preserves scope; pacing alone and actual pending E1 remain negative',()=>{ + const e=current(),pick=choice(e);expect(pick?.index).toBe(1); + expect(hasNativePostAnswerCeoPosture({status:'ready',calls:[retry.mode,retry.pacing],assistantMessages:[]},'SCOPE EXPANSION',pattern,retry.selectedAt,[])).toBe(false); + expect(retry.pendingProposal.answered).toBe(false); + expect(ceoExpansionPacingReady('next screen',e.transcript,pick!,[])).toBe(false); + }); + test('the preserving option can be reordered or use equivalent individual-walkthrough wording',()=>{ + const e=current(),q=e.pacing.questions[0]!;q.options.reverse(); + q.options[2]!.label='All proposals individually'; + q.options[2]!.description='Each proposal separately with Add / Defer / Skip / Hold. No item is skipped or merged without your approval. Delays the remaining review.'; + e.visible=pane(e.pacing as NativePlanQuestionCall,0);expect(choice(e)?.index).toBe(3); + }); + test.each(['foreign','unanswered mode','wrong mode','already answered','mixed packet','mismatched viewport','narrowing','bundled approval','quoted assurance','duplicate preserving choice','multiple pending calls'])('%s cannot authorize pacing',kind=>{ + const e=current(),q=e.pacing.questions[0]!,o=q.options[0]!; + if(kind==='foreign')e.pacing.sessionId='foreign'; + if(kind==='unanswered mode')e.transcript.calls[0]!.answered=false; + if(kind==='wrong mode')e.transcript.calls[0]!.answers={[e.transcript.calls[0]!.questions[0]!.question]:'HOLD SCOPE'}; + if(kind==='already answered')e.pacing.answered=true; + if(kind==='mixed packet')e.pacing.questions.push({...structuredClone(q),header:'Extra scope',question:'Approve all proposals now?'}); + if(kind==='narrowing')o.description+=' Add E1 and drop E2 now.'; + if(kind==='bundled approval')o.label='Full per-item split and approve all'; + if(kind==='quoted assurance')o.description=o.description.replace('No proposal is dropped or merged without your say','"No proposal is dropped or merged without your say"'); + if(kind==='duplicate preserving choice')q.options[1]=structuredClone(o); + if(kind==='multiple pending calls')e.transcript.calls.push({...structuredClone(e.pacing),toolUseId:'another-pending-call'}); + if(kind!=='mismatched viewport')e.visible=pane(e.pacing as NativePlanQuestionCall,0); + else e.visible=e.visible.replace('Full per-item split','Narrow first'); + if(['foreign','unanswered mode','wrong mode','already answered'].includes(kind))expect(choice(e)).toBeNull(); + else expect(choice(e)?.index).toBe(0); + }); + test('the pacing transition needs its successful bound ACK and a different current pane',()=>{ + const e=current(),pick=choice(e)!;e.transcript.calls[1]=structuredClone(retry.pacing); + const c=e.transcript.calls[1]!,events:NativePublicToolEvent[]=[ + {kind:'use',name:'AskUserQuestion',sessionId:c.sessionId,toolUseId:c.toolUseId,timestamp:new Date(Date.parse(c.answeredAt!)-1000).toISOString(),input:{questions:c.questions}}, + {kind:'result',sessionId:c.sessionId,toolUseId:c.toolUseId,timestamp:c.answeredAt!,isError:false}, + ]; + // Request time is synthetic; the captured ACK time and request body are retained. + expect(ceoExpansionPacingReady('a different current pane',e.transcript,pick,events)).toBe(true); + expect(ceoExpansionPacingReady(e.visible,e.transcript,pick,events)).toBe(false); + expect(ceoExpansionPacingReady('a different current pane',e.transcript,pick,events.slice(0,1))).toBe(false); + events[1]!.isError=true;expect(ceoExpansionPacingReady('a different current pane',e.transcript,pick,events)).toBe(false); + events[1]!.isError=false;c.answers={[c.questions[0]!.question]:c.questions[0]!.options[1]!.label}; + expect(ceoExpansionPacingReady('a different current pane',e.transcript,pick,events)).toBe(false); + }); + function acknowledgedProposal() { + // Derived transition only: pending E1 never received an actual paid ACK. + // Missing original request times below are explicitly synthetic. + const mode=structuredClone(retry.mode),proposal=structuredClone(retry.pendingProposal) as NativePlanQuestionCall; + proposal.answered=true;proposal.answers={[proposal.questions[0]!.question]:proposal.questions[0]!.options[0]!.label};proposal.unansweredQuestionIndices=[]; + proposal.answeredAt=new Date(Date.parse(retry.pacing.answeredAt)+2000).toISOString(); + const transcript={status:'ready' as const,calls:[mode,proposal],assistantMessages:[]}; + const events:NativePublicToolEvent[]=transcript.calls.flatMap(c=>[ + {kind:'use' as const,name:'AskUserQuestion',sessionId:c.sessionId,toolUseId:c.toolUseId,timestamp:new Date(Date.parse(c.answeredAt!)-1000).toISOString(),input:{questions:c.questions}}, + {kind:'result' as const,sessionId:c.sessionId,toolUseId:c.toolUseId,timestamp:c.answeredAt!,isError:false}, + ]); + return {transcript,events}; + } + test('a separately acknowledged current proposal establishes scope expansion through its real before/after comparison',()=>{ + const e=acknowledgedProposal();expect(hasNativePostAnswerCeoPosture(e.transcript,'SCOPE EXPANSION',pattern,retry.selectedAt,e.events)).toBe(true); + expect(hasNativePostAnswerCeoPosture(e.transcript,'SCOPE EXPANSION',/cathedral/i,retry.selectedAt,e.events)).toBe(false); + }); + test.each(['ordinal/source link','decimal decision identity','before/after paraphrase','defer','skip'])('%s preserves the same current proposal',kind=>{ + const e=acknowledgedProposal(),c=e.transcript.calls[1]!,q=c.questions[0]!; + if(kind==='ordinal/source link')q.question=q.question.replace('E1: Project-shared views (ledger row S1)','Proposal 1 of 7: E1 — Project-shared views [source](PLAN.md)'); + if(kind==='decimal decision identity')q.question=q.question.replace('D3.1 —','D12.3.1 —'); + if(kind==='before/after paraphrase')q.question=q.question.replace('Today the plan saves a view for one member only. E1 adds','As written, each member keeps private views. E1 would introduce'); + c.answers={[q.question]:q.options[kind==='defer'?1:kind==='skip'?2:0]!.label};e.events[2]!.input={questions:c.questions}; + expect(hasNativePostAnswerCeoPosture(e.transcript,'SCOPE EXPANSION',pattern,retry.selectedAt,e.events)).toBe(true); + }); + test.each(['pending','missing ACK','wrong proposal identity','no current baseline','vague baseline','second question','quoted comparison','foreign','selected pause'])('%s supplies no proposal completion',kind=>{ + const e=acknowledgedProposal(),c=e.transcript.calls[1]!,q=c.questions[0]!; + if(kind==='pending')c.answered=false; + if(kind==='missing ACK')e.events.pop(); + if(kind==='wrong proposal identity')q.question=q.question.replace('E1 adds','E2 adds'); + if(kind==='no current baseline')q.question=q.question.replace('Today the plan saves','Previously an unrelated plan saved'); + if(kind==='vague baseline')q.question=q.question.replace('Today the plan saves a view for one member only.','Today the plan is interesting.'); + if(kind==='second question')q.question=q.question.replace('ELI10:','ELI10: Should we remove access checks?'); + if(kind==='quoted comparison')q.question=q.question.replace('ELI10: Today','ELI10: "Today').replace('Stakes if','"\nStakes if'); + if(kind==='foreign')c.sessionId='foreign'; + c.answers={[q.question]:q.options[kind==='selected pause'?3:0]!.label};e.events[2]!.input={questions:c.questions}; + expect(hasNativePostAnswerCeoPosture(e.transcript,'SCOPE EXPANSION',pattern,retry.selectedAt,e.events)).toBe(false); + }); +}); +describe('complete candidate split is navigation with an actual ACK boundary',()=>{ +const f=completeInventory; +const actualFrame=f.viewport; +function state(){const pacing=structuredClone(f.pacing);pacing.answered=false;delete pacing.answers;delete pacing.answeredAt;delete pacing.unansweredQuestionIndices;return{pacing,transcript:{status:'ready' as const,calls:[structuredClone(f.mode),pacing],assistantMessages:[]}};} +function pane(c:any){const q=c.questions[0];return ['☐ '+q.header,q.question,...q.options.map((o:any,i:number)=>`${i?' ':'❯'} ${i+1}. ${o.label}`),'4. Type something.','5. Chat about this','Enter to select · ↑/↓ to navigate · Esc to cancel'].join('\n');} +const verify=(name:string,pass:boolean)=>test(name,()=>expect(pass).toBe(true)); +const choose=(e=state(),screen=pane(e.pacing))=>ceoExpansionPacingChoice(screen,e.transcript,f.selectedAt); +verify('actual retained frame selects the complete seven-candidate walkthrough',choose(state(),actualFrame)?.index===1); +for(const [name,mutate]of Object.entries({ + 'eight complete candidates':(q:any)=>{q.question=q.question.replaceAll('7 expansion candidates','8 expansion candidates').replace('7 adjacent improvements','8 adjacent improvements').replace('E7 cross-project views.','E7 cross-project views, E8 shared pinned groups.').replaceAll('Seven','Eight');q.options[0].label=q.options[0].label.replace('7 questions','8 questions');q.options[0].description=q.options[0].description.replace('E7','E8');}, + 'different proposal prefix':(q:any)=>{q.question=q.question.replace(/\bE(?=\d)/g,'P');q.options.forEach((o:any)=>{o.description=o.description.replace(/\bE(?=\d)/g,'P');});}, + 'complete walkthrough label':(q:any)=>{q.options[0].label='A: Complete walkthrough, 7 questions (recommended)';}, + 'one per item with explicit range':(q:any)=>{q.options[0].description='One question per item, E1 to E7.';}, + 'reordered choices':(q:any)=>{q.options.reverse();}, +})){const e=state();mutate(e.pacing.questions[0]);verify(name,choose(e)?.index===(name==='reordered choices'?3:1));} +for(const [name,mutate]of Object.entries({ + 'partial range':(q:any)=>{q.options[0].description=q.options[0].description.replace('E7','E6');}, + 'wrong number of questions':(q:any)=>{q.options[0].label=q.options[0].label.replace('7','6');}, + 'wrong declared count':(q:any)=>{q.question=q.question.replace('7 expansion candidates','8 expansion candidates');}, + 'missing candidate':(q:any)=>{q.question=q.question.replace(', E7 cross-project views','');}, + 'duplicate candidate':(q:any)=>{q.question=q.question.replace('E7 cross-project views','E6 cross-project views');}, + 'mixed proposal IDs':(q:any)=>{q.question=q.question.replace('E7 cross-project views','P7 cross-project views');}, + 'narrow selected walk':(q:any)=>{q.options[0].description+=' Except E4.';}, + 'selected scope approval':(q:any)=>{q.options[0].description+=' Approve E1 immediately.';}, + 'selected deletion':(q:any)=>{q.options[0].label+=' and delete E7';}, + 'selected grouping':(q:any)=>{q.options[0].description+=' Batch E1 and E2 together.';}, + 'quoted only range':(q:any)=>{q.options[0].description='"'+q.options[0].description+'"';}, + 'code-only range':(q:any)=>{q.options[0].description='`'+q.options[0].description+'`';}, + 'negated complete walk':(q:any)=>{q.options[0].label=q.options[0].label.replace('Full split','Not a full split');}, + 'duplicate complete choice':(q:any)=>{q.options[1]=structuredClone(q.options[0]);}, + 'extra question':(q:any)=>{q.question=q.question.replace('ELI10:','ELI10: Should all candidates ship?');}, + 'unconditional approval':(q:any)=>{q.question=q.question.replace('ELI10:','ELI10: This answer approves every expansion.');}, + 'quoted whole-question approval':(q:any)=>{q.question=q.question.replace('ELI10:','ELI10: “Choosing Full split approves E1 immediately.”');}, + 'hidden universal effect in another option':(q:any)=>{q.question=q.question.replace('B) Narrow first:','B) Regardless of choice, approve E1. Narrow first:');}, + 'historical inventory':(q:any)=>{q.question=q.question.replace('The delight scan produced','Previously the delight scan produced');}, + 'fenced brief':(q:any)=>{q.question='```\n'+q.question+'\n```';}, + 'missing comparison marker':(q:any)=>{q.question=q.question.replace('Note: options differ in kind, not coverage — no completeness score.','');}, +})){const e=state();mutate(e.pacing.questions[0]);verify(name,choose(e)?.index!==1);} +for(const [name,mutate]of Object.entries({ + 'foreign session':(e:any)=>{e.pacing.sessionId='foreign';}, + 'unanswered mode':(e:any)=>{e.transcript.calls[0].answered=false;}, + 'already answered pacing':(e:any)=>{e.pacing.answered=true;}, + 'mixed question packet':(e:any)=>{e.pacing.questions.push({...structuredClone(e.pacing.questions[0]),header:'Extra',question:'Approve everything?'});}, + 'another pending call':(e:any)=>{e.transcript.calls.push({...structuredClone(e.pacing),toolUseId:'other'});}, +})){const e=state();mutate(e);verify(name,choose(e)?.index!==1);} +const e=state(),choice=choose(e,actualFrame)!;const acknowledged={status:'ready' as const,calls:[f.mode,f.pacing,f.pending],assistantMessages:[]}; +const actualNext=f.nextViewport; +verify('actual pacing ACK and different pending E1 pane complete navigation',ceoExpansionPacingReady(actualNext,acknowledged,choice,f.publicEvents)); +verify('intended key without actual ACK does not complete navigation',!ceoExpansionPacingReady(actualNext,e.transcript,choice,f.publicEvents)); +verify('missing result does not complete navigation',!ceoExpansionPacingReady(actualNext,acknowledged,choice,f.publicEvents.filter((e:any)=>e.kind!=='result'))); +verify('failed result does not complete navigation',!ceoExpansionPacingReady(actualNext,acknowledged,choice,f.publicEvents.map((e:any)=>({...e,isError:e.kind==='result'})))); +verify('same old pane does not complete navigation',!ceoExpansionPacingReady(actualFrame,acknowledged,choice,f.publicEvents)); +verify('pacing and pending E1 supply no completed posture',!hasNativePostAnswerCeoPosture(acknowledged,'SCOPE EXPANSION',/expansion|10x|delight|dream/i,f.selectedAt,f.publicEvents)); +}); + +describe('candidate inventory cannot approve scope',()=>{ +const f=completeInventory; +function state(){const pacing=structuredClone(f.pacing);pacing.answered=false;delete pacing.answers;delete pacing.answeredAt;delete pacing.unansweredQuestionIndices;return{pacing,transcript:{status:'ready' as const,calls:[structuredClone(f.mode),pacing],assistantMessages:[]}};} +function pane(c:any){const q=c.questions[0];return ['☐ '+q.header,q.question,...q.options.map((o:any,i:number)=>`${i?' ':'❯'} ${i+1}. ${o.label}`),'4. Type something.','5. Chat about this','Enter to select · ↑/↓ to navigate · Esc to cancel'].join('\n');} +const mutations={ + 'inventory actor grants all candidates':(q:any)=>q.question=q.question.replace('E7 cross-project views.','E7 cross-project views; we approve all seven now.'), + 'inventory item claims current approval':(q:any)=>q.question=q.question.replace('E7 cross-project views.','E7 cross-project views (already approved).'), + 'inventory item has bare approval status':(q:any)=>q.question=q.question.replace('E7 cross-project views.','E7 cross-project views (approved).'), + 'inventory all items are approved':(q:any)=>q.question=q.question.replace('E7 cross-project views.','E7 cross-project views; all seven are approved.'), + 'inventory imperative ship grant':(q:any)=>q.question=q.question.replace('E7 cross-project views.','E7 cross-project views; ship all seven now.'), + 'inventory scope disposition':(q:any)=>q.question=q.question.replace('E7 cross-project views.','E7 cross-project views; all seven are in scope.'), + 'inventory skipped candidate':(q:any)=>q.question=q.question.replace('E7 cross-project views.','E7 cross-project views (deferred).'), + 'title claims inventory approved':(q:any)=>q.question=q.question.replace('How do you want to decide them?','All seven are already approved. How do you want to decide them?'), + 'rationale claims candidates in scope':(q:any)=>q.question=q.question.replace('The delight scan produced','All candidates are in scope. The delight scan produced'), + 'rationale claims prior approval':(q:any)=>q.question=q.question.replace('The delight scan produced','These items have been approved. The delight scan produced'), +}; +for (const [name,mutate] of Object.entries(mutations)) test(name,()=>{ + const e=state();mutate(e.pacing.questions[0]); + expect(ceoExpansionPacingChoice(pane(e.pacing),e.transcript,f.selectedAt)?.index).not.toBe(1); +}); +test('descriptive Update and delete feature titles remain supported',()=>{ + expect(ceoExpansionPacingChoice(f.viewport,state().transcript,f.selectedAt)?.index).toBe(1); +}); +}); +describe('complete per-proposal pacing preserves every candidate without granting scope',()=>{ + const f=nativePacing77.completePerProposal; + function state(){const transcript=structuredClone(f.transcript);return{transcript,pacing:transcript.calls.at(-1)!};} + function pane(c:any){const q=c.questions[0];return ['☐ '+q.header,q.question,...q.options.map((o:any,i:number)=>`${i?' ':'❯'} ${i+1}. ${o.label}`),'4. Type something.','5. Chat about this','Enter to select · ↑/↓ to navigate · Esc to cancel'].join('\n');} + function choose(e=state(),screen=pane(e.pacing)){return ceoExpansionPacingChoice(screen,e.transcript as any,f.selectionStartedAt);} + test('actual parenthesized full inventory binds one question per proposal',()=>{ + const e=state(),choice=choose(e,f.viewport)!; + expect(choice?.index).toBe(1); + expect(ceoExpansionPacingReady('Next proposal',e.transcript as any,choice,f.events as any)).toBe(false); + expect(hasNativePostAnswerCeoPosture(e.transcript as any,'SCOPE EXPANSION',/expansion|10x|delight|dream/i,f.selectionStartedAt,f.events as any)).toBe(false); + }); + const positive={ + 'different complete inventory prefix':(q:any)=>{q.question=q.question.replace(/\bP(?=\d)/g,'E');}, + 'numeric and word counts':(q:any)=>{q.question=q.question.replace('Seven expansion','7 expansion').replace('7 independent','seven independent');}, + 'colon-delimited independent inventory':(q:any)=>{q.question=q.question.replace('expansions (','expansions: ').replace('inline rename).','inline rename.');}, + 'different card identity':(q:any)=>{q.question=q.question.replace('D4.0','D12.0');}, + 'reordered choices':(q:any)=>{q.options.reverse();}, + 'no quoted task context':(q:any)=>{q.question=q.question.replace(' on "Add saved project views"','');}, + 'candidate terminology':(q:any)=>{q.options[0].label=q.options[0].label.replace('per proposal','per candidate');q.options[0].description=q.options[0].description.replace('Every proposal','Every candidate');}, + }; + for(const [name,mutate] of Object.entries(positive))test(name,()=>{const e=state();mutate(e.pacing.questions[0]);expect(choose(e)?.index).toBe(name==='reordered choices'?3:1);}); + const negative={ + 'missing inventory item':(q:any)=>{q.question=q.question.replace(', P7 quick switcher + inline rename','');}, + 'duplicate item':(q:any)=>{q.question=q.question.replace('P7 quick switcher','P6 quick switcher');}, + 'mixed prefixes':(q:any)=>{q.question=q.question.replace('P7 quick switcher','E7 quick switcher');}, + 'wrong title count':(q:any)=>{q.question=q.question.replace('Seven expansion','Eight expansion');}, + 'wrong described question count':(q:any)=>{q.question=q.question.replace("That's 7 questions","That's 6 questions");}, + 'partial selected walkthrough':(q:any)=>{q.options[0].description=q.options[0].description.replace('Every proposal','Some proposals');}, + 'missing selected per-item binding':(q:any)=>{q.options[0].label=q.options[0].label.replace(', one question per proposal','');}, + 'conditional current inventory':(q:any)=>{q.question=q.question.replace('I have 7','If I have 7');}, + 'historical inventory':(q:any)=>{q.question=q.question.replace('I have 7','Previously I had 7');}, + 'quoted mapping':(q:any)=>{q.options[0].label='A) Full split (recommended)';q.options[0].description='"One question per proposal. Every proposal gets its own Add / Defer / Skip / Hold."';}, + 'code-only mapping':(q:any)=>{q.options[0].description='`'+q.options[0].description+'`';}, + 'negated full split':(q:any)=>{q.options[0].label=q.options[0].label.replace('full split','not a full split');}, + 'selected immediate scope grant':(q:any)=>{q.options[0].description+=' We approve P1 now.';}, + 'universal approval in another option':(q:any)=>{q.options[1].description+=' Regardless of choice, approve P1 now.';}, + 'hidden inventory grant':(q:any)=>{q.question=q.question.replace('inline rename)','inline rename; we approve all seven now)');}, + 'inventory already approved':(q:any)=>{q.question=q.question.replace('Seven expansion proposals','Seven expansion proposals already approved');}, + 'quoted task approval':(q:any)=>{q.question=q.question.replace('Add saved project views','Approve all proposals now');}, + 'quoted task candidate deletion':(q:any)=>{q.question=q.question.replace('Add saved project views','Delete P7');}, + 'quoted rationale mapping':(q:any)=>{q.question=q.question.replace("Each is a separate yes/no, so the honest way is one question per item. That's 7 questions plus a final confirmation.","\"Each is a separate yes/no, so the honest way is one question per item. That's 7 questions plus a final confirmation.\"");}, + 'scope grant after task title':(q:any)=>{q.question=q.question.replace('views".','views"; approve P1 now.');}, + 'disguised omission assurance':(q:any)=>{q.options[0].description=q.options[0].description.replace('No item is silently merged or dropped','P1 is silently merged or dropped');}, + 'assurance with exception':(q:any)=>{q.options[0].description+=' Except P4.';}, + 'narrowing assurance':(q:any)=>{q.options[0].description+=' No item outside the top three is included.';}, + 'batch selected proposals':(q:any)=>{q.options[0].description+=' Batch P1 and P2 together.';}, + 'duplicate full choice':(q:any)=>{q.options[1]=structuredClone(q.options[0]);}, + }; + for(const [name,mutate] of Object.entries(negative))test(name,()=>{const e=state();mutate(e.pacing.questions[0]);expect(choose(e)?.index).not.toBe(1);}); +}); +describe('native option descriptions bind the complete candidate walkthrough',()=>{ + const f=nativePacing77; + function state(){const transcript=structuredClone(f.transcript);return{transcript,pacing:transcript.calls.at(-1)!};} + function pane(c:any){const q=c.questions[0];return ['☐ '+q.header,q.question,...q.options.map((o:any,i:number)=>`${i?' ':'❯'} ${i+1}. ${o.label}`),'4. Type something.','5. Chat about this','Enter to select · ↑/↓ to navigate · Esc to cancel'].join('\n');} + function choose(e=state(),screen=pane(e.pacing)){return ceoExpansionPacingChoice(screen,e.transcript as any,f.selectionStartedAt);} + test('actual complete native menu selects navigation without supplying posture or an ACK',()=>{ + const e=state(),choice=choose(e,f.viewport)!; + expect(choice?.index).toBe(1); + expect(ceoExpansionPacingReady('Next proposal',e.transcript as any,choice,f.events as any)).toBe(false); + expect(hasNativePostAnswerCeoPosture(e.transcript as any,'SCOPE EXPANSION',/expansion|10x|delight|dream/i,f.selectionStartedAt,f.events as any)).toBe(false); + }); + const positive={ + 'numeric count presentation':(q:any)=>{q.question=q.question.replaceAll('Eight','8').replaceAll('eight','8');q.options[0].description=q.options[0].description.replaceAll('Eight','8');}, + 'mixed word and numeric counts':(q:any)=>{q.question=q.question.replace('Eight expansion','8 expansion');q.options[0].description=q.options[0].description.replace('Eight sequential','8 sequential');}, + 'different complete candidate prefix':(q:any)=>{q.question=q.question.replace(/\bE(?=\d)/g,'P');}, + 'different question chain identity':(q:any)=>{q.question=q.question.replace('D4.0','D12.0');q.options[0].description=q.options[0].description.replaceAll('D4.','D12.');}, + 'reordered native choices':(q:any)=>{q.options.reverse();}, + 'explicit candidate range without duplicated option prose':(q:any)=>{q.options[0].label='A: Full split, 8 questions (recommended)';q.options[0].description='One question per candidate, E1 through E8.';}, + 'one per proposal label':(q:any)=>{q.options[0].label=q.options[0].label.replace('one per item','one per proposal');}, + 'no prior approach annotation':(q:any)=>{q.question=q.question.replace(', approach C approved','');}, + }; + for(const [name,mutate] of Object.entries(positive))test(name,()=>{ + const e=state();mutate(e.pacing.questions[0]);expect(choose(e)?.index).toBe(name==='reordered native choices'?3:1); + }); + const negative={ + 'hyphenated larger count cannot be read as its last digit':(q:any)=>{q.question=q.question.replaceAll('Eight','Twenty-eight').replaceAll('eight','twenty-eight');q.options[0].description=q.options[0].description.replaceAll('Eight','Twenty-eight');}, + 'spaced larger count cannot be read as its last digit':(q:any)=>{q.question=q.question.replaceAll('Eight','Twenty eight').replaceAll('eight','twenty eight');q.options[0].description=q.options[0].description.replaceAll('Eight','Twenty eight');}, + 'unsupported tens in title are not a single count':(q:any)=>{q.question=q.question.replace('Eight expansion','Thirty eight expansion');}, + 'unsupported tens in inventory are not a single count':(q:any)=>{q.question=q.question.replace('eight candidates:','forty eight candidates:');}, + 'unsupported tens in sequence are not a single count':(q:any)=>{q.options[0].description=q.options[0].description.replace('Eight sequential','Ninety eight sequential');}, + 'conjoined cardinal is not its last component':(q:any)=>{q.question=q.question.replace('Eight expansion','One hundred and eight expansion');}, + 'wrong title count':(q:any)=>{q.question=q.question.replace('Eight expansion','Seven expansion');}, + 'wrong inventory count':(q:any)=>{q.question=q.question.replace('eight candidates:','seven candidates:');}, + 'missing candidate':(q:any)=>{q.question=q.question.replace(', E8 views feeding digests/dashboards','');}, + 'duplicate candidate':(q:any)=>{q.question=q.question.replace('E8 views feeding','E7 views feeding');}, + 'foreign candidate prefix':(q:any)=>{q.question=q.question.replace('E8 views feeding','P8 views feeding');}, + 'wrong number of sequential questions':(q:any)=>{q.options[0].description=q.options[0].description.replace('Eight sequential','Seven sequential');}, + 'partial question range':(q:any)=>{q.options[0].description=q.options[0].description.replace('D4.8','D4.7');}, + 'late range start':(q:any)=>{q.options[0].description=q.options[0].description.replace('D4.1','D4.2');}, + 'foreign question chain':(q:any)=>{q.options[0].description=q.options[0].description.replaceAll('D4.','D5.');}, + 'additional question chain':(q:any)=>{q.options[0].description+=' Then D5.1.';}, + 'wrong label count':(q:any)=>{q.options[0].label=q.options[0].label.replace('one per item','7 questions');}, + 'no per-item label':(q:any)=>{q.options[0].label='A: Full split (recommended)';}, + 'quoted sequential range':(q:any)=>{q.options[0].description='"'+q.options[0].description+'"';}, + 'code-only sequential range':(q:any)=>{q.options[0].description='`'+q.options[0].description+'`';}, + 'conditional complete inventory':(q:any)=>{q.question=q.question.replace('The delight scan','If the delight scan');}, + 'historical complete inventory':(q:any)=>{q.question=q.question.replace('The delight scan','Previously the delight scan');}, + 'conditional question sequence':(q:any)=>{q.options[0].description='If approved, '+q.options[0].description;}, + 'historical question sequence':(q:any)=>{q.options[0].description='Previously: '+q.options[0].description;}, + 'negated complete choice':(q:any)=>{q.options[0].label='A: Not a full split, one per item';}, + 'sequence correction':(q:any)=>{q.options[0].description+=' Correction: Stop after four questions.';}, + 'selected scope approval':(q:any)=>{q.options[0].description+=' Approve E1 immediately.';}, + 'selected candidate omission':(q:any)=>{q.options[0].description+=' Except E4.';}, + 'selected merging action':(q:any)=>{q.options[0].description+=' Merge E1 and E2.';}, + 'unconditional omission':(q:any)=>{q.options[0].description=q.options[0].description.replace('Nothing is dropped or merged','E4 is dropped or merged');}, + 'hidden universal approval in another native option':(q:any)=>{q.options[1].description+=' Regardless of choice, approve E1 immediately.';}, + 'common candidate approval':(q:any)=>{q.question=q.question.replace('ELI10:','ELI10: This answer approves every expansion.');}, + 'current inventory approval':(q:any)=>{q.question=q.question.replace('E8 views feeding digests/dashboards.','E8 views feeding digests/dashboards (approved).');}, + 'approval in source context':(q:any)=>{q.question=q.question.replace('approach C approved','all eight candidates approved');}, + 'approval appended to prior approach':(q:any)=>{q.question=q.question.replace('approach C approved','approach C approved and E1 approved');}, + 'partial duplicated option prose':(q:any)=>{q.question=q.question.replace('Net:','A) Full split\nNet:');}, + 'duplicate complete choice':(q:any)=>{q.options[1]=structuredClone(q.options[0]);}, + 'extra question':(q:any)=>{q.question=q.question.replace('ELI10:','ELI10: Should we ship every item?');}, + }; + for(const [name,mutate] of Object.entries(negative))test(name,()=>{ + const e=state();mutate(e.pacing.questions[0]);expect(choose(e)?.index).not.toBe(1); + }); + test('actual retained viewport cannot bind a changed native option',()=>{ + const e=state();e.pacing.questions[0]!.options[0]!.label='A: Other menu';expect(choose(e,f.viewport)?.index).not.toBe(1); + }); +}); + + +describe('counted native per-item menu is pacing, not a substantive approval',()=>{ + const f=nativePacing77.countedNativeB955; + function state(){const mode=structuredClone(f.mode),pacing=structuredClone(f.pacing);pacing.answered=false;delete pacing.answers;delete pacing.answeredAt;delete pacing.unansweredQuestionIndices;return{mode,pacing,transcript:{status:'ready' as const,calls:[mode,pacing],assistantMessages:[]}};} + const screen=(c:any)=>pane(c,0); + const choose=(e=state(),visible=screen(e.pacing))=>ceoExpansionPacingChoice(visible,e.transcript as any,f.selectedAt); + test('complete captured native packet and observed display preserve the substantive allowance',()=>{ + const e=state(); + expect(choose(e)?.index).toBe(1); + expect(choose(e,f.viewport)?.index).toBe(1); + const pick=choose(e)!; + const next={status:'ready' as const,calls:[structuredClone(f.mode),structuredClone(f.pacing),structuredClone(f.pending)],assistantMessages:[]}; + const events=f.events.map(v=>v.kind==='use'?{...v,input:{questions:next.calls.find(c=>c.toolUseId===v.toolUseId)!.questions}}:v) as NativePublicToolEvent[]; + expect(ceoExpansionPacingReady(f.nextViewport,next as any,pick,events)).toBe(true); + expect(hasNativePostAnswerCeoPosture(next as any,'SCOPE EXPANSION',pattern,f.selectedAt,events)).toBe(false); + expect(nextCeoPostureContinuation(f.nextViewport,next as any,'SCOPE EXPANSION',f.selectedAt,new Set(),false)).toBe('question'); + expect(nextCeoPostureContinuation(f.nextViewport,next as any,'SCOPE EXPANSION',f.selectedAt,new Set(),true)).toBeNull(); + expect(f.pending.answered).toBe(false); + }); + const positive={ + 'question wording describes pacing intent':(q:any)=>{q.question=q.question.replace('Eleven expansion proposals: full per-item chain, narrow first, or batch?','How should we present the eleven expansion proposals: individually or in batches?');}, + 'explicit numeric count and independent candidate terminology':(q:any)=>{q.question=q.question.replace('Eleven expansion proposals','11 expansion candidates').replace('11 independent add-ons','eleven independent candidates');}, + 'proposals can name natural add/remove changes':(q:any)=>{q.question=q.question.replace('E1 shared visibility','E1 add shared views').replace('E2 versioned payload','E2 remove duplicate controls');}, + 'another complete set of explicit identities':(q:any)=>{q.question=q.question.replace(/\bE(?=\d)/g,'P').replaceAll('L3','Q9');}, + 'consistent reordered comparison and native options':(q:any)=>{q.options.reverse();}, + 'native labels carry letters too':(q:any)=>{q.options.forEach((o:any,i:number)=>{o.label=String.fromCharCode(65+i)+') '+o.label;});}, + }; + for(const[name,change]of Object.entries(positive))test(name,()=>{const e=state();change(e.pacing.questions[0]);expect(choose(e)?.index).toBe(name.startsWith('consistent reordered')?3:1);}); + const negative={ + 'missing declared candidate':(q:any)=>{q.question=q.question.replace(', L3 auto-persist last filters','');}, + 'duplicate declared identity':(q:any)=>{q.question=q.question.replace('L3 auto-persist last filters','E10 auto-persist last filters');}, + 'wrong title count':(q:any)=>{q.question=q.question.replace('Eleven expansion','Twelve expansion');}, + 'wrong question count in selected option':(q:any)=>{q.options[0].description=q.options[0].description.replace('11 per-item','10 per-item');}, + 'wrong rationale question count':(q:any)=>{q.question=q.question.replace('11 short questions','10 short questions');}, + 'partial per-item mapping':(q:any)=>{q.question=q.question.replace('Each needs its own','Some need their own');}, + 'another option owns the complete selected comparison':(q:any)=>{q.question=q.question.replace('A) Proceed with the full split (recommended)','A) Approve the first proposal (recommended)');}, + 'selected option lacks its own comparison':(q:any)=>{q.question=q.question.replace('✅ You see and rule on all 11 proposals; none are cut by me before you weigh in','');}, + 'quoted mapping is not evidence':(q:any)=>{q.options[0].description='"'+q.options[0].description+'"';}, + 'historical inventory':(q:any)=>{q.question=q.question.replace('The 10x analysis produced','Previously the 10x analysis produced');}, + 'conditional inventory':(q:any)=>{q.question=q.question.replace('The 10x analysis produced','If the 10x analysis produced');}, + 'inventory asserts approved status':(q:any)=>{q.question=q.question.replace('L3 auto-persist last filters','L3 auto-persist last filters (already approved)');}, + 'inventory conceals an actor grant':(q:any)=>{q.question=q.question.replace('L3 auto-persist last filters','L3 auto-persist last filters; we approve all eleven now');}, + 'inventory caption imperatively approves':(q:any)=>{q.question=q.question.replace('E1 shared visibility','E1 approve all proposals');}, + 'inventory caption declares approved':(q:any)=>{q.question=q.question.replace('E1 shared visibility','E1 approved shared views');}, + 'inventory caption defers other items':(q:any)=>{q.question=q.question.replace('E1 shared visibility','E1 defer others');}, + 'inventory caption hides imperative after a noun':(q:any)=>{q.question=q.question.replace('E1 shared visibility','E1 shared visibility and approve E2');}, + 'selected immediate scope approval':(q:any)=>{q.options[0].description+=' Approve E1 now.';}, + 'selected implicit approval':(q:any)=>{q.options[0].description+=' All proposals are included.';}, + 'selected omission':(q:any)=>{q.options[0].description+=' Except E4.';}, + 'selected grouping':(q:any)=>{q.options[0].description+=' Batch E1 and E2 together.';}, + 'unconditional effect in an unselected option':(q:any)=>{q.options[1].description+=' Regardless of choice, include E1 now.';}, + 'grant concealed in task title':(q:any)=>{q.question=q.question.replace('Add saved project views','Approve all proposals now');}, + 'extra decision':(q:any)=>{q.question=q.question.replace('ELI10:','ELI10: Should we remove access checks?');}, + 'duplicate preserving option':(q:any)=>{q.options[1]=structuredClone(q.options[0]);}, + }; + for(const[name,change]of Object.entries(negative))test(name,()=>{const e=state();change(e.pacing.questions[0]);expect(choose(e)?.index).not.toBe(1);}); + test('descriptive inventory nouns remain valid and substantive scope cards remain substantive',()=>{ + const e=state();e.pacing.questions[0]!.question=e.pacing.questions[0]!.question.replace('E1 shared visibility','E1 delete history views');expect(choose(e)?.index).toBe(1); + const pending=structuredClone(f.pending),transcript={status:'ready' as const,calls:[structuredClone(f.mode),pending],assistantMessages:[]}; + expect(ceoExpansionPacingChoice(screen(pending),transcript as any,f.selectedAt)).toBeNull(); + pending.questions[0]!.question=pending.questions[0]!.question.replace(/^D3\.1[^\n]+/,'D3.1 — Should we split the shared-view proposal into separate schemas?'); + expect(ceoExpansionPacingChoice(screen(pending),transcript as any,f.selectedAt)).toBeNull(); + }); + test('mode ownership, matching pane and actual ACK remain mandatory',()=>{ + const e=state();e.pacing.sessionId='foreign';expect(choose(e)).toBeNull(); + const noMode=state();noMode.mode.answered=false;expect(choose(noMode)).toBeNull(); + const ack=state(),pick=choose(ack)!;expect(pick?.index).toBe(1); + expect(ceoExpansionPacingReady('next',ack.transcript as any,pick,[])).toBe(false); + }); +}); + + +describe('same-proposal discussion control makes no scope decision',()=>{ + const f=nativePacing77.countedNativeB955; + function state(){ + const mode=structuredClone(f.mode),proposal=structuredClone(f.pending) as NativePlanQuestionCall; + // The actual proposal stayed pending. This derived ACK exercises only the + // downstream predicate; it cannot convert the original paid timeout to PASS. + proposal.answered=true;proposal.unansweredQuestionIndices=[]; + proposal.answers={[proposal.questions[0]!.question]:proposal.questions[0]!.options[0]!.label}; + proposal.answeredAt='2026-09-15T20:44:00.000Z'; + const calls=[mode,proposal]; + const events=f.events.filter(e=>calls.some(c=>c.toolUseId===e.toolUseId)).map(e=>e.kind==='use'?{...e,input:{questions:calls.find(c=>c.toolUseId===e.toolUseId)!.questions}}:{...e}) as NativePublicToolEvent[]; + events.push({kind:'result',sessionId:proposal.sessionId,toolUseId:proposal.toolUseId,timestamp:proposal.answeredAt,isError:false}); + return{proposal,transcript:{status:'ready' as const,calls,assistantMessages:[]},events}; + } + const matches=(e=state())=>hasNativePostAnswerCeoPosture(e.transcript,'SCOPE EXPANSION',pattern,f.selectedAt,e.events); + test('actual stop-and-discuss current E1 content remains nonoperative under a synthetic Include ACK',()=>{expect(f.pending.answered).toBe(false);expect(matches()).toBe(true);}); + test.each(['Pause the review. Discuss E1 before proceeding.','Discuss E1 before continuing; stop the chain.'])('equivalent two-clause procedural control: %s',description=>{ + const e=state();e.proposal.questions[0]!.options[3]!.description=description;expect(matches(e)).toBe(true); + }); + test.each(['Stop the chain; discuss E2 before continuing.','Stop the chain; approve E1 before continuing.','Stop the chain; discuss E1 before continuing. Add E2.', + 'Discuss E1 before continuing.','Stop the chain.','"Stop the chain; discuss E1 before continuing."','Previously stop the chain; discuss E1 before continuing.', + 'If needed, stop the chain; discuss E1 before continuing.','Stop the chain; discuss E1 before implementing it.'])('foreign, incomplete or operative control stays negative: %s',description=>{ + const e=state();e.proposal.questions[0]!.options[3]!.description=description;expect(matches(e)).toBe(false); + }); + test('pending, selected Hold, duplicate and foreign ACKs still supply no posture',()=>{ + for(const change of [ + (e:ReturnType)=>{e.proposal.answered=false;}, + (e:ReturnType)=>{const q=e.proposal.questions[0]!;e.proposal.answers={[q.question]:q.options[3]!.label};}, + (e:ReturnType)=>{e.events.push({...e.events.at(-1)!});}, + (e:ReturnType)=>{e.events.at(-1)!.sessionId='foreign';}, + ]){const e=state();change(e);expect(matches(e)).toBe(false);} + }); +}); + + +test.each(['acknowledged pacing','missing pacing ACK'])('actual paid posture loop preserves the substantive allowance: %s',async scenario=>{ + const f=nativePacing77.countedNativeB955; + const source=fs.readFileSync(path.join(import.meta.dir,'skill-e2e-plan-ceo-mode-routing.test.ts'),'utf8'); + const planDeclaration=source.match(/^const PLAN = \[[\s\S]*?^\]\.join\('\\n'\);/m)?.[0]; + expect(planDeclaration).toBeDefined(); + const plan=new Function(`${planDeclaration}; return PLAN;`)(); + const start=source.indexOf(' const budgetMs = 240_000;'),end=source.indexOf(" outcome = 'posture_confirmed';",start); + expect(start).toBeGreaterThan(0);expect(end).toBeGreaterThan(start); + const loop=source.slice(start,end+" outcome = 'posture_confirmed';".length); + const keys=['Bun','Date','c','session','sincePick','selectionStartedAt','question','fixture','capture','readPlanCountTranscript', + 'readPendingQuestion','hasNativePostAnswerCeoPosture','ceoModeSubmissionInput','ceoExpansionPacingReady','ceoExpansionPacingChoice', + 'nextCeoPostureContinuation','capturePlanCountQuestion','planCountQuestionInput','selectPtyNumberedOption','isPlanReadyVisible','isNumberedOptionListVisible', + 'EXPANSION_PACING_CALLS','modeIndex','artifacts','visibleAtMode','postureSource']; + const compiled=new Bun.Transpiler({loader:'ts'}).transformSync(`async function run(b){const {${keys.join(',')}}=b;let outcome;${loop};return {outcome,continuedQuestion,pacingCalls};}`); + const run=new Function(compiled+';return run;')(); + const pending=structuredClone(f.pacing);pending.answered=false;delete pending.answers;delete pending.answeredAt;delete pending.unansweredQuestionIndices; + const proposal=structuredClone(f.pending) as NativePlanQuestionCall; + let stage=0,clock=f.selectedAt; + const sends:string[]=[]; + const snapshots:string[]=[]; + const view=()=>stage===0?f.viewport:f.nextViewport; + const session={hermeticConfigDir:'fixture-native',pendingQuestionFile:'fixture-pending',exited:()=>false,exitCode:()=>null, + currentScreen:async()=>view(),visibleSince:()=>view(),visibleText:()=>view(),send:(value:string)=>{ + sends.push(value);stage++; + if(stage===2){proposal.answered=true;proposal.answers={[proposal.questions[0]!.question]:proposal.questions[0]!.options[0]!.label}; + proposal.answeredAt='2026-09-15T20:44:00.000Z';proposal.unansweredQuestionIndices=[];} + }}; + const readPlanCountTranscript=(_config:string,_cwd:string,emit:(e:NativePublicToolEvent)=>void)=>{ + const pacing=stage===0||scenario==='missing pacing ACK'?pending:f.pacing; + const calls=stage===0?[f.mode,pacing]:[f.mode,pacing,proposal]; + const events=f.events.filter(e=>calls.some(c=>c.toolUseId===e.toolUseId)&&!(e.kind==='result'&&e.toolUseId===f.pacing.toolUseId&&!pacing.answered)) + .map(e=>e.kind==='use'?{...e,input:{questions:calls.find(c=>c.toolUseId===e.toolUseId)!.questions}}:{...e}) as NativePublicToolEvent[]; + if(proposal.answered)events.push({kind:'result',sessionId:proposal.sessionId,toolUseId:proposal.toolUseId,timestamp:proposal.answeredAt!,isError:false}); + events.forEach(emit);return{status:'ready',calls,assistantMessages:[]}; + }; + const bindings={Bun:{sleep:async(ms:number)=>{clock+=ms;}},Date:{now:()=>clock},c:{mode:'SCOPE EXPANSION',postureRe:pattern},session,sincePick:0, + selectionStartedAt:f.selectedAt,question:{nativeCall:f.mode},fixture:{cwd:'fixture-root'},capture:(state:string)=>snapshots.push(state),readPlanCountTranscript, + readPendingQuestion:()=>undefined,hasNativePostAnswerCeoPosture,ceoModeSubmissionInput,ceoExpansionPacingReady,ceoExpansionPacingChoice,nextCeoPostureContinuation, + capturePlanCountQuestion,planCountQuestionInput,selectPtyNumberedOption:async(s:any,index:number)=>s.send(String(index)),isPlanReadyVisible,isNumberedOptionListVisible, + EXPANSION_PACING_CALLS:1,modeIndex:2,artifacts:{},visibleAtMode:'captured mode menu', + postureSource:{path:path.join('fixture-root','PLAN.md'),content:plan}}; + if(scenario==='missing pacing ACK')await expect(run(bindings)).rejects.toThrow('no posture match'); + else expect(await run(bindings)).toEqual({outcome:'posture_confirmed',continuedQuestion:true,pacingCalls:1}); + expect(sends).toEqual(scenario==='missing pacing ACK'?['1']:['1','1']); + expect(snapshots.length).toBeGreaterThan(0); + expect(f.pending.answered).toBe(false); // Final synthetic ACK is never paid evidence. +}); +}); + +describe('ceo-mode-posture-ad', () => { +const captured = captured_ceo_mode_posture_ad; +const patterns = { + 'HOLD SCOPE': /\b(rigor|bulletproof|hold\s*scope|maximum\s+rigor)\b/i, + 'SCOPE EXPANSION': /\b(expansion|10x|delight|dream|cathedral|opt[\s-]?in)\b/i, +}; +function replay(index: number, change?: (rows: any[]) => void) { + const item = captured.cases[index]!; + const rows = structuredClone(item.records); + change?.(rows); + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'ceo-posture-ad-')); + const project = path.join(dir, 'projects', 'owned'); + fs.mkdirSync(project, {recursive:true}); + fs.writeFileSync(path.join(project, item.process.sessionId+'.jsonl'), rows.map(row=>JSON.stringify(row)).join('\n')+'\n'); + const events: NativePublicToolEvent[]=[]; + try { return {item, transcript:readPlanCountTranscript(dir,item.process.cwd,event=>events.push(event)),events}; } + finally { fs.rmSync(dir,{recursive:true,force:true}); } +} +function matches(e: ReturnType) { + const mode=e.item.mode as keyof typeof patterns; + return hasNativePostAnswerCeoPosture(e.transcript,mode,patterns[mode],e.item.selectedAt,e.events); +} +function rebind(e: ReturnType) { + const decision=e.transcript.calls[1]!; + e.events[2]!.input={questions:decision.questions}; + decision.answers={[decision.questions[0]!.question]:decision.questions[0]!.options[0]!.label}; +} + +for (const index of [0,1]) describe(`${captured.cases[index]!.mode} actual completed mode application`,()=>{ + test('the exact mode answer and concrete scope decision supply posture without finalized prose',()=>{ + const e=replay(index); + expect(e.item.actualFailure.state).toBe('failed'); + expect(e.transcript.calls.map(call=>call.toolUseId)).toEqual([e.item.modeToolUseId,e.item.decisionToolUseId]); + expect(nativeCeoModeAnswer(e.transcript,e.item.mode as keyof typeof patterns,e.item.selectedAt)?.toolUseId).toBe(e.item.modeToolUseId); + expect(e.transcript.assistantMessages.every(message=>Date.parse(message.timestamp){ + const e=replay(index);const [mode,decision]=e.transcript.calls;const q=decision!.questions[0]!; + switch(failure){ + case 'wrong selected mode':mode!.answers![mode!.questions[0]!.question]=index===0?'SCOPE EXPANSION':'HOLD SCOPE';break; + case 'pending mode':mode!.answered=false;break; + case 'failed mode':mode!.failed=true;break; + case 'answer before selection':mode!.answeredAt=new Date(e.item.selectedAt-1).toISOString();break; + case 'pending decision':decision!.answered=false;break; + case 'failed decision':decision!.failed=true;break; + case 'foreign session':decision!.sessionId=e.events[2]!.sessionId=e.events[3]!.sessionId='foreign';break; + case 'pre-mode request':e.events[2]!.timestamp=e.events[0]!.timestamp;break; + case 'reply before request':decision!.answeredAt=e.events[3]!.timestamp=new Date(Date.parse(e.events[2]!.timestamp)-1).toISOString();break; + case 'reply before selection':decision!.answeredAt=e.events[3]!.timestamp=new Date(e.item.selectedAt-1).toISOString();break; + case 'reply timestamp mismatch':e.events[3]!.timestamp=new Date(Date.parse(decision!.answeredAt!)+1).toISOString();break; + case 'wrong tool':e.events[2]!.name='Read';break; + case 'missing request':e.events.splice(2,1);break; + case 'missing reply':e.events.splice(3,1);break; + case 'failed public reply':e.events[3]!.isError=true;break; + case 'duplicate request':e.events.push({...e.events[2]!});break; + case 'duplicate reply':e.events.push({...e.events[3]!});break; + case 'request mismatch':e.events[2]!.input={questions:[]};break; + case 'unknown answer':decision!.answers![q.question]='Unrecognized';break; + case 'extra question':decision!.questions.push({...structuredClone(q),header:'Also',question:'Also remove the CI gate?'});rebind(e);break; + case 'extra option':q.options.push({label:'Remove the CI gate',description:'A separate obligation.'});rebind(e);break; + case 'multiselect':q.multiSelect=true;rebind(e);break; + case 'quoted decision':q.question=q.question.split('\n').map(line=>'> '+line).join('\n');rebind(e);break; + case 'fenced decision':q.question='```text\n'+q.question+'\n```';rebind(e);break; + case 'mere mode mention':q.question=`D6 — Continue the review?\nSelected ${e.item.mode}.`;rebind(e);break; + case 'extra obligation':q.question+=' Also, should we remove the CI gate?';rebind(e);break; + case 'extra imperative':q.question+=' Also remove the CI gate.';rebind(e);break; + case 'option imperative':q.options[0]!.description+=' Please remove the CI gate.';rebind(e);break; + } + expect(matches(e),failure).toBe(false); + }); + test.each(['Delete the CI gate.', 'Ship the new endpoint now.', 'After that, disable authentication.'])('an instruction appended after the final comparison is not part of the scope brief: %s', extra=>{ + const e=replay(index);const q=e.transcript.calls[1]!.questions[0]!; + q.question+=' '+extra;rebind(e);expect(matches(e)).toBe(false); + }); + test('foreign, sidechain, missing and failed native records do not become completed evidence',()=>{ + for(const change of [(rows:any[])=>{rows[3].cwd='/foreign';},(rows:any[])=>{rows[3].isSidechain=true;}, + (rows:any[])=>{rows.pop();},(rows:any[])=>{rows[4].message.content[0].is_error=true;}]) expect(matches(replay(index,change))).toBe(false); + }); +}); + +test('HOLD requires the explicit out-of-scope deferral and its selected defer answer',()=>{ + for(const change of [(q:any)=>{q.question=q.question.replace('Under HOLD SCOPE, keep or defer','Under HOLD SCOPE, automatically add');}, + (q:any)=>{q.question=q.question.replace('not in the plan text','required by the plan text');}, + (q:any)=>{q.question=q.question.replace('pure additions, not repairs to meet a stated invariant','repairs needed to meet a stated invariant');}, + (q:any)=>{q.options[0].label='Keep all three (recommended)';}, + (q:any)=>{q.options[1].label='Remove CI gate';}]){ + const e=replay(0);change(e.transcript.calls[1]!.questions[0]);rebind(e);expect(matches(e)).toBe(false); + } + const e=replay(0);const q=e.transcript.calls[1]!.questions[0]!; + e.transcript.calls[1]!.answers={[q.question]:q.options[1]!.label};expect(matches(e)).toBe(false); +}); + +test('completed expansion decisions require application of the selected mode',()=>{ + for(const change of [(q:any)=>{q.question='D6 — Continue the review?\nSelected SCOPE EXPANSION.';}, + (q:any)=>{q.question=q.question.replace('SCOPE EXPANSION mode','SELECTIVE EXPANSION mode');}, + (q:any)=>{q.question=q.question.replace('SCOPE EXPANSION mode','HOLD SCOPE mode');}, + (q:any)=>{q.options[1].label='Enable telemetry';}]){ + const e=replay(1);change(e.transcript.calls[1]!.questions[0]);rebind(e);expect(matches(e)).toBe(false); + } +}); +}); + +describe('ceo-prerequisite-ad-v2', () => { +const captured = captured_ceo_prerequisite_ad_v2; +function pending(){const c=structuredClone(captured.completedCall) as NativePlanQuestionCall;c.answered=false;delete c.answers;delete c.answeredAt;delete c.unansweredQuestionIndices;return c;} +// Native identities and questions are exact; pending panes are synthetic projections. +function pane(c:NativePlanQuestionCall,index:number){const q=c.questions[index]!;return [ + c.questions.length>1?'← '+c.questions.map((v,i)=>`${i`${i?' ':'❯'} ${i+1}. ${v.label}`), + `Enter to select · ${c.questions.length>1?'Tab/Arrow keys':'↑/↓'} to navigate · Esc to cancel`].join('\n');} +function frame(c:NativePlanQuestionCall,index:number){const visible=pane(c,index);return {visible,active:capturePlanCountQuestion(visible,new Set(),0,true,c)!,routing:nativePlanCallFingerprint(c,0,true)};} +test('AD v2 actual comma prerequisite selects standard review on its active native tab',()=>{ + const actual=captured.completedCall,q=actual.questions[2]!; + expect(actual.answered).toBe(true);expect(actual.failed).toBe(false);expect(actual.answers[q.question]).toBe('Run /office-hours now'); + const c=pending(),f=frame(c,2);expect(f.active.nativeQuestionIndex).toBe(2); + expect(planCountPrerequisitePick(f.routing,f.active)).toBe(2); + const a=nextCeoModeNavigation(f.visible,'HOLD SCOPE',new Set(),c);expect(a.kind).toBe('question'); + if(a.kind==='question')expect(planCountQuestionInput(f.visible,a.question,a.index)).toBe('2'); +}); + +test('AD v2 prerequisite presentation and actual order do not choose the action',()=>{ + for(const header of ['Office hours','Design doc','Prerequisite'])for(const reverse of [false,true]){ + const c=pending();c.questions[2]!.header=header;c.questions[2]!.question=c.questions[2]!.question.replace(/^D3 — /,'D41: '); + if(reverse)c.questions[2]!.options.reverse();const f=frame(c,2); + expect(planCountPrerequisitePick(f.routing,f.active)).toBe(reverse?1:2); + for(const index of [0,1]){const other=frame(c,index);expect(planCountPrerequisitePick(other.routing,other.active)).toBeNull();} + } + const c=pending();c.questions=[c.questions[2]!];let f=frame(c,0);expect(planCountPrerequisitePick(f.routing,f.active)).toBe(2); + c.questions[0]!.question='Run /office-hours now or proceed with standard review?\nNo design doc exists for the current feature. The scoped review can begin on the supplied plan.'; + c.questions[0]!.options[0]!.description='Create the design document first; then resume standard review.'; + for(const description of ['Proceed with standard review.','Proceed straight to Step 0 of the review.']){ + c.questions[0]!.options[1]!.description=description;f=frame(c,0);expect(planCountPrerequisitePick(f.routing,f.active)).toBe(2); + } +}); + +test('AD v2 prerequisite declines no other task or conditional action',()=>{ + const changes:Array<(c:NativePlanQuestionCall)=>void>=[ + c=>{c.questions[2]!.question=c.questions[2]!.question.replace(/^.*\n/,'Should we deploy the feature now?\n');}, + c=>{c.questions[2]!.question='Example: '+c.questions[2]!.question;}, + c=>{c.questions[2]!.question='> '+c.questions[2]!.question;}, + c=>{c.questions[2]!.question='```text\n'+c.questions[2]!.question+'\n```';}, + c=>{c.questions[2]!.question=c.questions[2]!.question.replace('Run /office-hours first, or proceed with the standard review?','Should we remove authorization? Run /office-hours first, or proceed with the standard review?');}, + c=>{c.questions[2]!.question+=' Approve production deployment?';}, + c=>{c.questions[2]!.question+=' You must run /office-hours first.';}, + c=>{c.questions[2]!.question+=' Standard review is forbidden until /office-hours completes.';}, + c=>{c.questions[2]!.options[1]!.label+=' if the tests pass';}, + c=>{c.questions[2]!.options[0]!.label+=' and rewrite the API';}, + c=>{c.questions[2]!.options[1]!.description='Proceed with standard review after completing /office-hours.';}, + c=>{c.questions[2]!.options[1]!.description='No review will run.';}, + c=>{c.questions[2]!.options[1]!.description='Proceed directly to Step 0 of the CEO review. Remove CI.';}, + c=>{c.questions[2]!.options[0]!.description='Do not run /office-hours.';}, + c=>{c.questions[2]!.options[0]!.description='Build a design doc first, then resume the review. Deploy to production.';}, + c=>{c.questions[2]!.options[1]!.description='';}, + c=>{c.questions[2]!.options.push({label:'Approve deployment'});}, + c=>{c.questions[2]!.multiSelect=true;}, + ]; + for(const change of changes){const c=pending();change(c);const f=frame(c,2);expect(planCountPrerequisitePick(f.routing,f.active)).toBeNull();} +}); + +test('AD v2 prerequisite requires the active native packet identity',()=>{ + const c=pending(),f=frame(c,2); + for(const active of [{...f.active,preReview:false},{...f.active,signature:'foreign:tool:question:2'}, + {...f.active,nativeQuestionIndex:0},{...f.active,promptSnippet:'Unrelated question'}, + {...f.active,nativeCall:undefined},{...f.active,options:[...f.active.options].reverse()}]) + expect(planCountPrerequisitePick(f.routing,active)).toBeNull(); + expect(planCountPrerequisitePick({...f.active,nativeCall:undefined})).toBeNull(); + for(const delta of [{answered:true},{failed:true},{sessionId:''},{toolUseId:''}]){const call={...pending(),...delta};const x=frame(call,2);expect(planCountPrerequisitePick(x.routing,x.active)).toBeNull();} +}); +}); diff --git a/test/ceo-mode-posture-ad.test.ts b/test/ceo-mode-posture-ad.test.ts deleted file mode 100644 index 7d6355b4e..000000000 --- a/test/ceo-mode-posture-ad.test.ts +++ /dev/null @@ -1,117 +0,0 @@ -import { describe, expect, test } from 'bun:test'; -import * as fs from 'node:fs'; -import * as os from 'node:os'; -import * as path from 'node:path'; -import { hasNativePostAnswerCeoPosture, nativeCeoModeAnswer } from './helpers/ceo-mode-option'; -import { readPlanCountTranscript, type NativePublicToolEvent } from './helpers/plan-count-transcript'; -import { E2E_TOUCHFILES, selectTests } from './helpers/touchfiles'; -import captured from './fixtures/ceo-mode-posture-ad.json'; - -const patterns = { - 'HOLD SCOPE': /\b(rigor|bulletproof|hold\s*scope|maximum\s+rigor)\b/i, - 'SCOPE EXPANSION': /\b(expansion|10x|delight|dream|cathedral|opt[\s-]?in)\b/i, -}; -function replay(index: number, change?: (rows: any[]) => void) { - const item = captured.cases[index]!; - const rows = structuredClone(item.records); - change?.(rows); - const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'ceo-posture-ad-')); - const project = path.join(dir, 'projects', 'owned'); - fs.mkdirSync(project, {recursive:true}); - fs.writeFileSync(path.join(project, item.process.sessionId+'.jsonl'), rows.map(row=>JSON.stringify(row)).join('\n')+'\n'); - const events: NativePublicToolEvent[]=[]; - try { return {item, transcript:readPlanCountTranscript(dir,item.process.cwd,event=>events.push(event)),events}; } - finally { fs.rmSync(dir,{recursive:true,force:true}); } -} -function matches(e: ReturnType) { - const mode=e.item.mode as keyof typeof patterns; - return hasNativePostAnswerCeoPosture(e.transcript,mode,patterns[mode],e.item.selectedAt,e.events); -} -function rebind(e: ReturnType) { - const decision=e.transcript.calls[1]!; - e.events[2]!.input={questions:decision.questions}; - decision.answers={[decision.questions[0]!.question]:decision.questions[0]!.options[0]!.label}; -} - -for (const index of [0,1]) describe(`${captured.cases[index]!.mode} actual completed mode application`,()=>{ - test('the exact mode answer and concrete scope decision supply posture without finalized prose',()=>{ - const e=replay(index); - expect(e.item.actualFailure.state).toBe('failed'); - expect(e.transcript.calls.map(call=>call.toolUseId)).toEqual([e.item.modeToolUseId,e.item.decisionToolUseId]); - expect(nativeCeoModeAnswer(e.transcript,e.item.mode as keyof typeof patterns,e.item.selectedAt)?.toolUseId).toBe(e.item.modeToolUseId); - expect(e.transcript.assistantMessages.every(message=>Date.parse(message.timestamp){ - const e=replay(index);const [mode,decision]=e.transcript.calls;const q=decision!.questions[0]!; - switch(failure){ - case 'wrong selected mode':mode!.answers![mode!.questions[0]!.question]=index===0?'SCOPE EXPANSION':'HOLD SCOPE';break; - case 'pending mode':mode!.answered=false;break; - case 'failed mode':mode!.failed=true;break; - case 'answer before selection':mode!.answeredAt=new Date(e.item.selectedAt-1).toISOString();break; - case 'pending decision':decision!.answered=false;break; - case 'failed decision':decision!.failed=true;break; - case 'foreign session':decision!.sessionId=e.events[2]!.sessionId=e.events[3]!.sessionId='foreign';break; - case 'pre-mode request':e.events[2]!.timestamp=e.events[0]!.timestamp;break; - case 'reply before request':decision!.answeredAt=e.events[3]!.timestamp=new Date(Date.parse(e.events[2]!.timestamp)-1).toISOString();break; - case 'reply before selection':decision!.answeredAt=e.events[3]!.timestamp=new Date(e.item.selectedAt-1).toISOString();break; - case 'reply timestamp mismatch':e.events[3]!.timestamp=new Date(Date.parse(decision!.answeredAt!)+1).toISOString();break; - case 'wrong tool':e.events[2]!.name='Read';break; - case 'missing request':e.events.splice(2,1);break; - case 'missing reply':e.events.splice(3,1);break; - case 'failed public reply':e.events[3]!.isError=true;break; - case 'duplicate request':e.events.push({...e.events[2]!});break; - case 'duplicate reply':e.events.push({...e.events[3]!});break; - case 'request mismatch':e.events[2]!.input={questions:[]};break; - case 'unknown answer':decision!.answers![q.question]='Unrecognized';break; - case 'extra question':decision!.questions.push({...structuredClone(q),header:'Also',question:'Also remove the CI gate?'});rebind(e);break; - case 'extra option':q.options.push({label:'Remove the CI gate',description:'A separate obligation.'});rebind(e);break; - case 'multiselect':q.multiSelect=true;rebind(e);break; - case 'quoted decision':q.question=q.question.split('\n').map(line=>'> '+line).join('\n');rebind(e);break; - case 'fenced decision':q.question='```text\n'+q.question+'\n```';rebind(e);break; - case 'mere mode mention':q.question=`D6 — Continue the review?\nSelected ${e.item.mode}.`;rebind(e);break; - case 'extra obligation':q.question+=' Also, should we remove the CI gate?';rebind(e);break; - case 'extra imperative':q.question+=' Also remove the CI gate.';rebind(e);break; - case 'option imperative':q.options[0]!.description+=' Please remove the CI gate.';rebind(e);break; - } - expect(matches(e),failure).toBe(false); - }); - test.each(['Delete the CI gate.', 'Ship the new endpoint now.', 'After that, disable authentication.'])('an instruction appended after the final comparison is not part of the scope brief: %s', extra=>{ - const e=replay(index);const q=e.transcript.calls[1]!.questions[0]!; - q.question+=' '+extra;rebind(e);expect(matches(e)).toBe(false); - }); - test('foreign, sidechain, missing and failed native records do not become completed evidence',()=>{ - for(const change of [(rows:any[])=>{rows[3].cwd='/foreign';},(rows:any[])=>{rows[3].isSidechain=true;}, - (rows:any[])=>{rows.pop();},(rows:any[])=>{rows[4].message.content[0].is_error=true;}]) expect(matches(replay(index,change))).toBe(false); - }); -}); - -test('HOLD requires the explicit out-of-scope deferral and its selected defer answer',()=>{ - for(const change of [(q:any)=>{q.question=q.question.replace('Under HOLD SCOPE, keep or defer','Under HOLD SCOPE, automatically add');}, - (q:any)=>{q.question=q.question.replace('not in the plan text','required by the plan text');}, - (q:any)=>{q.question=q.question.replace('pure additions, not repairs to meet a stated invariant','repairs needed to meet a stated invariant');}, - (q:any)=>{q.options[0].label='Keep all three (recommended)';}, - (q:any)=>{q.options[1].label='Remove CI gate';}]){ - const e=replay(0);change(e.transcript.calls[1]!.questions[0]);rebind(e);expect(matches(e)).toBe(false); - } - const e=replay(0);const q=e.transcript.calls[1]!.questions[0]!; - e.transcript.calls[1]!.answers={[q.question]:q.options[1]!.label};expect(matches(e)).toBe(false); -}); - -test('completed expansion decisions require application of the selected mode',()=>{ - for(const change of [(q:any)=>{q.question='D6 — Continue the review?\nSelected SCOPE EXPANSION.';}, - (q:any)=>{q.question=q.question.replace('SCOPE EXPANSION mode','SELECTIVE EXPANSION mode');}, - (q:any)=>{q.question=q.question.replace('SCOPE EXPANSION mode','HOLD SCOPE mode');}, - (q:any)=>{q.options[1].label='Enable telemetry';}]){ - const e=replay(1);change(e.transcript.calls[1]!.questions[0]);rebind(e);expect(matches(e)).toBe(false); - } -}); - -test('the new replay controls and fixture select the actual periodic mode-routing caller',()=>{ - for(const file of ['test/ceo-mode-posture-ad.test.ts','test/fixtures/ceo-mode-posture-ad.json']) - expect(selectTests([file],E2E_TOUCHFILES,[]).selected).toEqual(['plan-ceo-mode-routing']); -}); diff --git a/test/ceo-prerequisite-ad-v2.test.ts b/test/ceo-prerequisite-ad-v2.test.ts deleted file mode 100644 index 51051fb23..000000000 --- a/test/ceo-prerequisite-ad-v2.test.ts +++ /dev/null @@ -1,75 +0,0 @@ -import {expect,test} from 'bun:test'; -import {capturePlanCountQuestion,nativePlanCallFingerprint,planCountPrerequisitePick,planCountQuestionInput} from './helpers/claude-pty-runner'; -import {nextCeoModeNavigation} from './helpers/ceo-mode-option'; -import type {NativePlanQuestionCall} from './helpers/plan-count-transcript'; -import captured from './fixtures/ceo-prerequisite-ad-v2.json'; -function pending(){const c=structuredClone(captured.completedCall) as NativePlanQuestionCall;c.answered=false;delete c.answers;delete c.answeredAt;delete c.unansweredQuestionIndices;return c;} -// Native identities and questions are exact; pending panes are synthetic projections. -function pane(c:NativePlanQuestionCall,index:number){const q=c.questions[index]!;return [ - c.questions.length>1?'← '+c.questions.map((v,i)=>`${i`${i?' ':'❯'} ${i+1}. ${v.label}`), - `Enter to select · ${c.questions.length>1?'Tab/Arrow keys':'↑/↓'} to navigate · Esc to cancel`].join('\n');} -function frame(c:NativePlanQuestionCall,index:number){const visible=pane(c,index);return {visible,active:capturePlanCountQuestion(visible,new Set(),0,true,c)!,routing:nativePlanCallFingerprint(c,0,true)};} -test('AD v2 actual comma prerequisite selects standard review on its active native tab',()=>{ - const actual=captured.completedCall,q=actual.questions[2]!; - expect(actual.answered).toBe(true);expect(actual.failed).toBe(false);expect(actual.answers[q.question]).toBe('Run /office-hours now'); - const c=pending(),f=frame(c,2);expect(f.active.nativeQuestionIndex).toBe(2); - expect(planCountPrerequisitePick(f.routing,f.active)).toBe(2); - const a=nextCeoModeNavigation(f.visible,'HOLD SCOPE',new Set(),c);expect(a.kind).toBe('question'); - if(a.kind==='question')expect(planCountQuestionInput(f.visible,a.question,a.index)).toBe('2'); -}); - -test('AD v2 prerequisite presentation and actual order do not choose the action',()=>{ - for(const header of ['Office hours','Design doc','Prerequisite'])for(const reverse of [false,true]){ - const c=pending();c.questions[2]!.header=header;c.questions[2]!.question=c.questions[2]!.question.replace(/^D3 — /,'D41: '); - if(reverse)c.questions[2]!.options.reverse();const f=frame(c,2); - expect(planCountPrerequisitePick(f.routing,f.active)).toBe(reverse?1:2); - for(const index of [0,1]){const other=frame(c,index);expect(planCountPrerequisitePick(other.routing,other.active)).toBeNull();} - } - const c=pending();c.questions=[c.questions[2]!];let f=frame(c,0);expect(planCountPrerequisitePick(f.routing,f.active)).toBe(2); - c.questions[0]!.question='Run /office-hours now or proceed with standard review?\nNo design doc exists for the current feature. The scoped review can begin on the supplied plan.'; - c.questions[0]!.options[0]!.description='Create the design document first; then resume standard review.'; - for(const description of ['Proceed with standard review.','Proceed straight to Step 0 of the review.']){ - c.questions[0]!.options[1]!.description=description;f=frame(c,0);expect(planCountPrerequisitePick(f.routing,f.active)).toBe(2); - } -}); - -test('AD v2 prerequisite declines no other task or conditional action',()=>{ - const changes:Array<(c:NativePlanQuestionCall)=>void>=[ - c=>{c.questions[2]!.question=c.questions[2]!.question.replace(/^.*\n/,'Should we deploy the feature now?\n');}, - c=>{c.questions[2]!.question='Example: '+c.questions[2]!.question;}, - c=>{c.questions[2]!.question='> '+c.questions[2]!.question;}, - c=>{c.questions[2]!.question='```text\n'+c.questions[2]!.question+'\n```';}, - c=>{c.questions[2]!.question=c.questions[2]!.question.replace('Run /office-hours first, or proceed with the standard review?','Should we remove authorization? Run /office-hours first, or proceed with the standard review?');}, - c=>{c.questions[2]!.question+=' Approve production deployment?';}, - c=>{c.questions[2]!.question+=' You must run /office-hours first.';}, - c=>{c.questions[2]!.question+=' Standard review is forbidden until /office-hours completes.';}, - c=>{c.questions[2]!.options[1]!.label+=' if the tests pass';}, - c=>{c.questions[2]!.options[0]!.label+=' and rewrite the API';}, - c=>{c.questions[2]!.options[1]!.description='Proceed with standard review after completing /office-hours.';}, - c=>{c.questions[2]!.options[1]!.description='No review will run.';}, - c=>{c.questions[2]!.options[1]!.description='Proceed directly to Step 0 of the CEO review. Remove CI.';}, - c=>{c.questions[2]!.options[0]!.description='Do not run /office-hours.';}, - c=>{c.questions[2]!.options[0]!.description='Build a design doc first, then resume the review. Deploy to production.';}, - c=>{c.questions[2]!.options[1]!.description='';}, - c=>{c.questions[2]!.options.push({label:'Approve deployment'});}, - c=>{c.questions[2]!.multiSelect=true;}, - ]; - for(const change of changes){const c=pending();change(c);const f=frame(c,2);expect(planCountPrerequisitePick(f.routing,f.active)).toBeNull();} -}); - -test('AD v2 prerequisite requires the active native packet identity',()=>{ - const c=pending(),f=frame(c,2); - for(const active of [{...f.active,preReview:false},{...f.active,signature:'foreign:tool:question:2'}, - {...f.active,nativeQuestionIndex:0},{...f.active,promptSnippet:'Unrelated question'}, - {...f.active,nativeCall:undefined},{...f.active,options:[...f.active.options].reverse()}]) - expect(planCountPrerequisitePick(f.routing,active)).toBeNull(); - expect(planCountPrerequisitePick({...f.active,nativeCall:undefined})).toBeNull(); - for(const delta of [{answered:true},{failed:true},{sessionId:''},{toolUseId:''}]){const call={...pending(),...delta};const x=frame(call,2);expect(planCountPrerequisitePick(x.routing,x.active)).toBeNull();} -}); - -import {E2E_TOUCHFILES,selectTests} from './helpers/touchfiles'; -test('AD v2 prerequisite regression selects its existing mode workflow',()=>{ - for(const file of ['test/ceo-prerequisite-ad-v2.test.ts','test/fixtures/ceo-prerequisite-ad-v2.json']) - expect(selectTests([file],E2E_TOUCHFILES,[]).selected).toEqual(['plan-ceo-mode-routing']); -}); diff --git a/test/ceo-section-loading-fixture.test.ts b/test/ceo-section-loading-fixture.test.ts index f9790dde0..47035e4a3 100644 --- a/test/ceo-section-loading-fixture.test.ts +++ b/test/ceo-section-loading-fixture.test.ts @@ -5,6 +5,14 @@ import { CEO_SECTION_CACHE_PLAN, hasStaleFillRaceFinding, } from './helpers/ceo-section-loading-fixture'; +import captured_sdk_columnar_af from './fixtures/sdk-columnar-af.json'; +import captured_sdk_compact_sequence_aj from './fixtures/sdk-compact-sequence-aj.json'; +import captured_sdk_order_b_ag from './fixtures/sdk-order-b-ag.json'; +import fs_sdk_ordered_schedule_ar from 'node:fs'; +import fixture_sdk_ordering_ae from './fixtures/sdk-ordering-ae.json'; +import captured_sdk_original_order_ai from './fixtures/sdk-original-order-ai.json'; +import fixture_sdk_schedule_continuation_ah from './fixtures/sdk-schedule-continuation-ah.json'; +import fixture_sdk_stale_table_ad_v3 from './fixtures/sdk-stale-table-ad-v3.json'; describe('future-reader vocabulary in the actual AA finding', () => { const report = require('node:fs').readFileSync(require('node:path').join(import.meta.dir, 'fixtures/ceo-section-aa-report.md'), 'utf8'); @@ -1025,3 +1033,855 @@ describe('section fixture rollout metrics retain final acceptance without an imp expect(CEO_SECTION_CACHE_PLAN).toContain('repository.read returns\n an immutable absent-result DTO for a missing record, never undefined'); }); }); + +describe('sdk-columnar-af', () => { +const captured = captured_sdk_columnar_af; +const evidence = () => captured.retryFinding + '\n\n' + captured.retrySchedule; +function replace(text: string, before: string, after: string) { + expect(text).toContain(before); + return text.replace(before, after); +} + +test('AF exact retry columnar schedule establishes a post-write stale reader', () => { + expect(hasStaleFillRaceFinding(captured.retryReport)).toBe(true); + expect(hasStaleFillRaceFinding(captured.firstGuardedEvidence)).toBe(false); +}); + +test('AF columnar evidence binds named actors, keys and distinct versions independently of their spelling', () => { + expect(hasStaleFillRaceFinding(evidence())).toBe(true); + const varied = evidence().replace(/\bR1\b/g, 'R4').replace(/\bR2\b/g, 'R8').replace(/\bW\b/g, 'W3') + .replace(/\bK\b/g, 'profileKey').replace(/\bv1\b/g, 'oldVersion').replace(/\bv2\b/g, 'newVersion') + .replace(/->/g, '→'); + expect(hasStaleFillRaceFinding(varied)).toBe(true); + expect(hasStaleFillRaceFinding(evidence().replace(/\bv2\b/g, 'v1'))).toBe(false); +}); + +test('AF every ordered operation and version witness is required', () => { + for (const [before, after] of [ + ['get(K) -> undefined', 'get(K) -> v1'], + ['await repository.read -> v1', 'await repository.read -> v2'], + ['await write commits v2', 'await write fails'], + ['delete(K) (no entry)', 'keep(K)'], + ['| returns |', '| still pending |'], + ['resume: set(K, v1); return v1', 'resume: set(K, v2); return v2'], + ['get(K) -> v1; return v1', 'get(K) -> v2; return v2'], + ['v1 STALE | v2', 'v1 STALE | v1'], + ['6 | resume:', '8 | resume:'], + ]) expect(hasStaleFillRaceFinding(replace(evidence(), before!, after!))).toBe(false); + for (let event = 1; event <= 7; event++) { + expect(hasStaleFillRaceFinding(evidence().split('\n').filter(line => !line.trim().startsWith(`${event} |`)).join('\n'))).toBe(false); + } +}); + +test('AF a different key, reader, write or column cannot lend ownership', () => { + for (const [before, after] of [ + ['cache[K] | DB[K]', 'cache[K] | DB[J]'], + ['delete(K) (no entry)', 'delete(J) (no entry)'], + ['set(K, v1)', 'set(J, v1)'], + ['get(K) -> v1; return v1', 'get(J) -> v1; return v1'], + ['R2 read (begins after W)', 'R1 read (begins after W)'], + ['R2 read (begins after W)', 'R2 read (begins after W2)'], + ['R2 began after W completed (t5)', 'R1 began after W completed (t5)'], + ['R2 began after W completed (t5)', 'R2 began before W completed (t5)'], + ['R2 began after W completed (t5)', 'R2 began after W completed (t6)'], + ['observes v1 for up to 30 s.', 'observes v2 for up to 30 s.'], + ]) expect(hasStaleFillRaceFinding(replace(evidence(), before!, after!))).toBe(false); +}); + +test('AF current declarative execution cannot borrow a conditional, negated or quoted schedule', () => { + for (const [before, after] of [ + ['await write commits v2', 'write might commit v2'], + ['resume: set(K, v1); return v1', 'resume: no set(K, v1); return v1'], + ['VIOLATION t7:', 'If VIOLATION t7:'], + ['VIOLATION t7:', 'Quoted VIOLATION t7:'], + ['observes v1 for up to 30 s.', 'observes v1 for up to 30 s.?'], + ]) expect(hasStaleFillRaceFinding(replace(evidence(), before!, after!))).toBe(false); +}); + +test('AF the same named finding and a real top-level fence own the schedule', () => { + const text = evidence(); + for (const value of [ + captured.retrySchedule, + text.replace('(F1 evidence)', '(F2 evidence)'), + text.replace('Schedule S1 below', 'Schedule S2 below'), + captured.retryFinding + '\n' + text, + text.split('\n').map(line => '> ' + line).join('\n'), + '````text\n' + text + '\n````', + text.replace('```\n t |', '```javascript\n t |'), + text.slice(0, text.lastIndexOf('```')), + 'Example:\n\n' + text, + captured.retryFinding + '\n\nTemplate:\n' + captured.retrySchedule, + ]) expect(hasStaleFillRaceFinding(value)).toBe(false); +}); + +test('AF an original-caller allowance cannot excuse a stale cache or later caller', () => { + const allowed = 'Allowed by contract: R1 itself returns v1 (read in progress when write committed).'; + for (const value of [ + replace(evidence(), allowed, 'Allowed by contract: R2 itself returns v1 (read in progress when write committed).'), + replace(evidence(), allowed, 'The stale-fill behavior is accepted.'), + replace(evidence(), allowed, 'There is no stale-fill race.'), + replace(evidence(), allowed, 'The trace is impossible.'), + evidence() + '\n\nThis is not a violation. No guard is required.', + ]) expect(hasStaleFillRaceFinding(value)).toBe(false); +}); +test('AF every same-row assessment and an unproven source frame remain authoritative', () => { + for (const [cell, value] of [ + [6, 'Rejected: there is no stale-fill race.'], + [5, 'The stale-fill behavior is accepted. No guard is required.'], + [6, 'Rejected: “There is no stale-fill race.”'], + [6, 'Rejected: "The stale-fill behavior is accepted. No guard is required."'], + ] as const) { + const cells = captured.retryFinding.split('|'); cells[cell] = value; + expect(hasStaleFillRaceFinding(cells.join('|') + '\n\n' + captured.retrySchedule)).toBe(false); + } + expect(hasStaleFillRaceFinding('An unproven hypothesis:\n\n' + evidence())).toBe(false); +}); +}); + +describe('sdk-compact-sequence-aj', () => { +const captured = captured_sdk_compact_sequence_aj; +const sequence = 'fill starts, write commits, write deletes (no-op), fill sets pre-commit v1, later read hits v1.'; +const report = captured.finding; + +test('recognizes the captured current original-plan sequence without borrowing the amended diagram', () => { + expect(hasStaleFillRaceFinding(report)).toBe(true); + expect(hasStaleFillRaceFinding(report.replaceAll('v1', 'snapshot_A'))).toBe(true); + expect(hasStaleFillRaceFinding(report.replace('Schedule Diagram 2b: ', ''))).toBe(true); +}); + +test('requires the ordered original fill, commit, invalidation, old cache value and same later value', () => { + for (const changed of [ + sequence.replace('fill starts, ', ''), + sequence.replace('write commits, ', ''), + sequence.replace('write deletes (no-op), ', ''), + sequence.replace('fill sets pre-commit v1, ', ''), + sequence.replace(', later read hits v1', ''), + sequence.replace('later read hits v1', 'later read hits v2'), + sequence.replace('pre-commit v1', 'post-commit v1'), + sequence.replace('fill starts, write commits', 'write commits, fill starts'), + sequence.replace('write deletes (no-op), fill sets pre-commit v1', 'fill sets pre-commit v1, write deletes (no-op)'), + sequence.replace('later read hits', 'another key later read hits'), + ]) expect(hasStaleFillRaceFinding(report.replace(sequence, changed))).toBe(false); + expect(hasStaleFillRaceFinding(report.replace('Original plan', 'Amended plan'))).toBe(false); + expect(hasStaleFillRaceFinding(report.replace(sequence, '"' + sequence + '"'))).toBe(false); +}); + +test('preserves accepted-staleness and explicit dismissal boundaries', () => { + for (const suffix of [ + 'This is not a gap; no guard is needed.', + 'This staleness is the accepted consistency model.', + 'This finding is withdrawn.', + 'F1 is rejected.', + ]) expect(hasStaleFillRaceFinding(report.trimEnd() + '\n\n' + suffix)).toBe(false); + expect(hasStaleFillRaceFinding(report.replace('fill starts', 'fill never starts'))).toBe(false); +}); + +test('source, quotes and hypothetical framing cannot supply current coverage', () => { + for (const text of [ + '```text\n' + report + '```', + report.split('\n').map(line => '> ' + line).join('\n'), + report.split('\n').map(line => ' ' + line).join('\n'), + '## Historical example\n\n' + report, + '## Quoted source\n\n' + report, + 'An unproven hypothesis.\n\n' + report, + 'The following is a hypothetical example.\n\n' + report, + ]) expect(hasStaleFillRaceFinding(text)).toBe(false); + expect(hasStaleFillRaceFinding('## Historical example\nOld material.\n\n## Current review\n' + report)).toBe(true); + expect(hasStaleFillRaceFinding(report + '\n## Unrelated issue\nF2 is rejected.')).toBe(true); +}); + +test('source framing remains attached to descendant registry headings', () => { + for (const prefix of [ + '## Copied material\nThe following subsections reproduce source examples, not current findings.\n\n', + '## Input material\nThe following sections quote historical examples.\n\n', + 'The following subsections reproduce source examples, not current findings.\n\n', + ]) expect(hasStaleFillRaceFinding(prefix + report)).toBe(false); + expect(hasStaleFillRaceFinding('## Source notes\nThe following material quotes historical examples.\n\n## Current findings\n' + report)).toBe(true); +}); + +test('same finding assessments retain identity across sections and unrelated findings', () => { + for (const suffix of [ + '## F1 assessment\nThis finding is withdrawn.', + '## Final assessment\nF1 is rejected.', + '## F2\nUnrelated issue accepted.\n\n## Final assessment\nF1 is dismissed.', + ]) expect(hasStaleFillRaceFinding(report + '\n\n' + suffix)).toBe(false); + for (const suffix of [ + '## F2 assessment\nThis finding is withdrawn.', + '## Final assessment\nF2 is rejected.', + '## Quoted source\nF1 is rejected.', + '## Source notes\nThe following subsections quote historical examples.\n\n### F1 assessment\nThis finding is withdrawn.', + ]) expect(hasStaleFillRaceFinding(report + '\n\n' + suffix)).toBe(true); +}); +}); + +describe('sdk-order-b-ag', () => { +const captured = captured_sdk_order_b_ag; +const compactFirst = () => `${captured.first.finding}\n\n${captured.first.heading}\n\`\`\`\n${captured.first.trace}\n\`\`\``; +const compactRetry = () => `${captured.retry.finding}\n\n${captured.retry.heading}\n\`\`\`\n${captured.retry.trace}\n\`\`\``; + +test('actual first completed report proves a later stale cache hit', () => { + expect(hasStaleFillRaceFinding(captured.first.report)).toBe(true); +}); + +test('isolated Order B proves a later stale cache hit', () => { + expect(hasStaleFillRaceFinding(compactFirst())).toBe(true); +}); + +test('actual retry and its explicit original-sketch override establish the unsafe execution', () => { + expect(hasStaleFillRaceFinding(captured.retry.report)).toBe(true); + expect(hasStaleFillRaceFinding(compactRetry())).toBe(true); +}); + +function replaceOnce(text: string, before: string, after: string): string { + expect(text.includes(before)).toBe(true); + return text.replace(before, after); +} + +test('original-caller return or flight joining alone cannot supply the later cache reader', () => { + for (const [before, after] of [ + [' Order B: R2 begins after t5 -> cache hit v1 VIOLATION (until TTL or next write)\n', ''], + ['Order B: R2 begins after t5', 'Order B: R1 begins after t5'], + ['Order B: R2 begins after t5', 'Order B: R2 begins before t3'], + ['Order B: R2 begins after t5', 'Order B: R2 begins after t2'], + ['cache hit v1 VIOLATION', 'fresh DB read v2'], + ['cache hit v1 VIOLATION', 'cache hit v2 SAFE'], + ]) expect(hasStaleFillRaceFinding(replaceOnce(compactFirst(), before!, after!))).toBe(false); +}); + +test('all read, commit, invalidation and late-fill operations retain shared key and version ownership', () => { + for (const [before, after] of [ + ['R1 readProfile(k)', 'R1 readProfile(other)'], + ['W writeProfile(k, v2)', 'W writeProfile(other, v2)'], + ['R2 readProfile(k)', 'R2 readProfile(other)'], + ['cache[k]', 'cache[other]'], + ['inflight[k]', 'inflight[other]'], + ['set(k, v1)', 'set(other, v1)'], + ['set(k, v1)', 'set(k, v2)'], + ['read resolves v1; set(k, v1)', 'read resolves v2; set(k, v1)'], + ['miss; flight f1; await read', 'cache hit v1; return'], + ['await write ... commit v2', 'await write ... abort'], + ['delete(k) no-op; return', 'delete(other) no-op; return'], + ['delete(k) no-op; return', 'write still pending'], + ['read resolves v1; set(k, v1)', 'read resolves v1; return to R1 only'], + ['v1 BAD | -', 'v2 SAFE | -'], + ]) expect(hasStaleFillRaceFinding(replaceOnce(compactFirst(), before!, after!))).toBe(false); +}); + +test('quoted, conditional and impossible schedules are not actual asserted execution', () => { + const report = compactFirst(); + for (const changed of [ + report.split('\n').map(line => `> ${line}`).join('\n'), + `\`\`\`markdown\n${report}\n\`\`\``, + `An unproven hypothesis:\n${report}`, + replaceOnce(report, 'Schedule below shows', 'An unproven hypothesis: Schedule below shows'), + replaceOnce(report, 'Order B: R2 begins', 'Order B: If R2 begins'), + replaceOnce(report, 'Order B: R2 begins', 'Order B: R2 never begins'), + replaceOnce(report, 'Order B: R2 begins after t5 -> cache hit v1 VIOLATION', 'Order B: R2 begins after t5 -> cache hit v1 VIOLATION?'), + report + '\nThis trace is impossible.', + report + '\n\nThe trace is impossible.', + ]) expect(hasStaleFillRaceFinding(changed)).toBe(false); +}); + +test('one finding owns the original trace and every same-row assessment', () => { + const report = compactFirst(); + for (const changed of [ + replaceOnce(report, 'Async schedule (F1)', 'Async schedule (F9)'), + replaceOnce(report, '| F1 |', '| F9 |'), + replaceOnce(report, 'Fills overlapping a write are not cached (bounded hit-rate cost, visible in metric)', 'There is no stale-fill race.'), + replaceOnce(report, 'Fills overlapping a write are not cached (bounded hit-rate cost, visible in metric)', 'Rejected: "There is no stale-fill race."'), + replaceOnce(report, 'D3: single-flight `invalidate(key)` before and after the write; invalidated fills never `set`; `fill_discarded` metric', 'The stale-fill behavior is accepted. No guard is required.'), + ]) expect(hasStaleFillRaceFinding(changed)).toBe(false); +}); + +test('retry amendment alone and unasserted original-sketch annotations cannot prove a stale fill', () => { + const original = 'Original sketch: step 6 fills v1 after step 4 → R2 hits v1 → VIOLATION (S1).'; + for (const replacement of [ + '', + `"${original}"`, + `> ${original}`, + `If ${original}`, + `Example: ${original}`, + original.replace('fills v1', 'does not fill v1'), + original.replace('VIOLATION (S1).', 'VIOLATION (S1)?'), + original.replace('fills v1', 'fills v2'), + original.replace('after step 4', 'before step 4'), + original.replace('after step 4', 'after step 3'), + original.replace('step 6 fills', 'step 7 fills'), + original.replace('R2 hits v1', 'R1 receives v1'), + original.replace('R2 hits v1', 'R2 hits v2'), + original.replace('(S1)', '(S9)'), + ]) expect(hasStaleFillRaceFinding(replaceOnce(compactRetry(), original, replacement))).toBe(false); + expect(hasStaleFillRaceFinding(compactRetry() + '\n\nThe trace is impossible.')).toBe(false); +}); + +test('retry original override is bound to the same actors, cancelled token and completed write', () => { + for (const [before, after] of [ + ['R1 read (began before commit)', 'R1 read (began after commit)'], + ['R2 read (began after W resolves)', 'R2 read (began before W resolves)'], + ['R2 read (began after W resolves)', 'R2 read (other key, began after W resolves)'], + ['invalidate: cancel t1, detach, delete', 'invalidate: cancel other, detach, delete'], + ['writeProfile resolves (write "complete")', 'writeProfile still pending'], + ['await repo.write → v2 committed', 'await repo.write → aborted'], + ['read resolves v1; t1✗ → no fill', 'read resolves v2; t1✗ → no fill'], + ['read resolves v1; t1✗ → no fill', 'read resolves v1; other✗ → no fill'], + ['S1: R1 misses, W commits and deletes, R1 fills stale v1, R2 hits v1.', 'S1: R1 misses, W commits and deletes, R1 fills stale v1, R1 receives v1.'], + ['### 4. Async schedule (F1)', '### 4. Async schedule (F9)'], + ]) expect(hasStaleFillRaceFinding(replaceOnce(compactRetry(), before!, after!))).toBe(false); +}); + +test('consistent actor, key, version and pending-identity renaming preserves each causal proof', () => { + const names: Record = { + R1: 'R7', R2: 'R8', W: 'W9', k: 'profile_key', v1: 'oldValue', v2: 'newValue', + f1: 'flight_old', f2: 'flight_new', t1: 'token_old', t2: 'token_new', + }; + for (const report of [compactFirst(), compactRetry()]) { + const renamed = report.replace(/\b(?:R1|R2|W|k|v1|v2|f1|f2|t1|t2)\b/g, token => names[token]!); + expect(hasStaleFillRaceFinding(renamed)).toBe(true); + } +}); + +test('current findings cannot borrow assertion authority from a hypothetical preceding frame', () => { + for (const report of [compactFirst(), compactRetry()]) { + for (const prefix of ['An unproven hypothesis.', 'Historical example only.', 'The following is a hypothetical example.']) { + expect(hasStaleFillRaceFinding(`${prefix}\n\n${report}`)).toBe(false); + } + expect(hasStaleFillRaceFinding(`## Prior example\nA completed historical illustration.\n\n## Current findings\n${report}`)).toBe(true); + } +}); +}); + +describe('sdk-ordered-schedule-ar', () => { +const fs = fs_sdk_ordered_schedule_ar; +const report = fs.readFileSync(new URL('./fixtures/sdk-ordered-schedule-ar.md', import.meta.url), 'utf8'); +const row = report.split('\n').find(line => line.startsWith('| F1 |'))!; +const schedule = 'Schedule: read misses, write commits and deletes (no-op), read resolves and stores the pre-write snapshot. A later read hits the stale value'; + +test('an actual review supplies the stale-fill ordering without a concurrency keyword', () => { + expect(row).toContain(schedule); + expect(row).not.toMatch(/\b(?:race|concurrent|in-flight|pending)\b/i); + expect(hasStaleFillRaceFinding(row)).toBe(true); + expect(hasStaleFillRaceFinding(report)).toBe(true); + expect(hasStaleFillRaceFinding(row.replace('Original sketch', 'Original wrapper'))).toBe(true); + expect(hasStaleFillRaceFinding(row.replace('pre-write snapshot', 'old value'))).toBe(true); + expect(hasStaleFillRaceFinding(row.replace('A later read', 'The subsequent read'))).toBe(true); + expect(hasStaleFillRaceFinding(row.replaceAll('"', ''))).toBe(true); +}); + +test('every operation and the stale value observed by a later read are required', () => { + for (const [from, to] of [ + ['read misses, ', ''], + ['write commits and deletes (no-op), ', ''], + ['write commits and deletes', 'write rolls back and deletes'], + ['write commits and deletes', 'write commits without deleting'], + ['read resolves and stores the pre-write snapshot', 'read resolves and skips the fill'], + ['read resolves and stores the pre-write snapshot', 'read resolves and stores the fresh snapshot'], + ['A later read hits the stale value', 'The original read returns its own pre-write snapshot'], + ['A later read hits the stale value', 'A later read hits the fresh value'], + ['write commits and deletes (no-op), read resolves and stores the pre-write snapshot', 'read resolves and stores the pre-write snapshot, write commits and deletes (no-op)'], + ['write commits and deletes (no-op), read resolves', 'write commits and deletes (no-op) | read resolves'], + ['read resolves and stores', 'another reader resolves and stores'], + ['write commits and deletes (no-op)', 'write commits and deletes another key'], + ]) { + expect(row).toContain(from); + expect(hasStaleFillRaceFinding(row.replace(from, to))).toBe(false); + } +}); + +test('copied, conditional, quoted and hypothetical schedules cannot supply current evidence', () => { + for (const text of [ + '> ' + row, + '```text\n' + row + '\n```', + '## Historical example\n' + row, + 'Source:\n' + row, + 'Earlier review:\n' + row, + row.replace('Original sketch fills', 'Original sketch source excerpt only: fills'), + row.replace('Original sketch fills', 'Original sketch from an earlier review fills'), + row.replace('Schedule:', '\nFinding F2. Schedule:'), + '## Source notes\nThe following material is copied from a template.\n' + row, + row.replace('Original sketch', 'Quoted original sketch'), + row.replace('Schedule: read misses', 'Schedule: if a read misses'), + row.replace('Schedule: read misses', 'Hypothetical schedule: read misses'), + row.replace('read resolves and stores', 'read never resolves and stores'), + row.replace(schedule, '"' + schedule + '"'), + row.replace(schedule, '`' + schedule + '`'), + row.replace('read resolves and stores the pre-write snapshot', '`read resolves and stores the pre-write snapshot`'), + row.replace('Flag flip mid-read has the same shape.', 'This sequence is impossible.'), + ]) expect(hasStaleFillRaceFinding(text)).toBe(false); +}); + +test('a current dismissal stays a dismissal even when the original schedule is complete', () => { + for (const suffix of [ + 'F1 is withdrawn.', + 'F1 is "withdrawn".', + 'F1 is “withdrawn”.', + 'F1 is rejected.', + 'This finding is dismissed.', + 'This is not a bug; no fix is needed.', + 'The stale-fill behavior is permitted.', + ]) expect(hasStaleFillRaceFinding(row + '\n\n' + suffix)).toBe(false); + expect(hasStaleFillRaceFinding(row + '\n\nF2 is rejected.')).toBe(true); + expect(hasStaleFillRaceFinding('## Historical example\nOld material.\n\n## Current findings\n' + row)).toBe(true); +}); +}); + +describe('sdk-ordering-ae', () => { +const fixture = fixture_sdk_ordering_ae; +const found = hasStaleFillRaceFinding; +const trace = fixture.f1.split('|')[4]!.trim(); +function withTrace(value: string): string { + const cells = fixture.f1.split('|'); + cells[4] = ` ${value} `; + return cells.join('|'); +} + +test('actual completed F1 report row supplies ordered stale-fill evidence without a race keyword', () => { + expect(found(fixture.f1)).toBe(true); + expect(fixture.provenance.historicalOutcome).toContain('timeout480032ms'); + expect(trace).not.toMatch(/\b(?:race|in-flight|concurrent|pending)\b/i); + expect(found(`F1 — P1: ${trace}`)).toBe(true); +}); + +test('ordering evidence requires miss, committed invalidation, stale refill and later stale readers', () => { + for (const value of [ + 'Reader fills the pre-commit snapshot; write commits and deletes; read misses; every later reader sees stale data.', + 'Read misses; reader then fills the pre-commit snapshot; write commits and deletes; every later reader sees stale data.', + 'Write commits and deletes; read misses; reader then fills the pre-commit snapshot; every later reader sees stale data.', + 'Read misses; reader then fills the pre-commit snapshot; every later reader sees stale data.', + 'Read misses, write commits; reader then fills the pre-commit snapshot; every later reader sees stale data.', + 'Read misses, write commits and deletes; every later reader sees stale data.', + 'Read misses, write commits and deletes; reader then fills the post-commit snapshot; every later reader sees fresh data.', + 'Read misses, write commits and deletes; the original reader returns its pre-commit snapshot to its own caller; every later reader sees fresh data.', + ]) expect(found(withTrace(value))).toBe(false); +}); + +test('explicit other cache, key or reader references cannot borrow the anonymous same-read trace', () => { + for (const value of [ + 'Read misses cache A, write commits and deletes cache B, reader then fills cache A with the pre-commit snapshot; every later reader sees stale data in cache A.', + 'Read misses key u1, write commits and deletes key u2, reader then fills key u1 with the pre-commit snapshot; every later reader sees stale data for key u1.', + 'Read R1 misses, write commits and deletes, reader R2 then fills the pre-commit snapshot; every later reader sees stale data.', + ]) expect(found(withTrace(value))).toBe(false); +}); + +test('hypothetical, negated and unestablished traces do not assert a current defect', () => { + for (const value of [ + `If ${trace[0]!.toLowerCase()}${trace.slice(1)}`, + `A hypothetical example: ${trace}`, + `An unproven hypothesis: ${trace}`, + `The following trace is impossible: ${trace}`, + `An unrelated illustration: ${trace}`, + `It is unclear whether this happens: ${trace}`, + `This trace did not occur: ${trace}`, + trace.replace('Read misses', 'Read may miss'), + trace.replace('write commits and deletes', 'write does not commit or delete'), + trace.replace('reader then fills', 'reader never fills'), + trace.replace('every later reader sees stale data', 'every later reader never sees stale data'), + 'Read misses, write commits and deletes, reader then fills the pre-commit snapshot; every later reader sees stale data?', + 'Read misses, write commits and deletes, reader then fills the pre-commit snapshot; every later reader sees stale data. This scenario is impossible.', + ]) expect(found(withTrace(value))).toBe(false); +}); + +test('copied source and independent rows or cells cannot supply missing ordered operations', () => { + for (const value of [`> ${fixture.f1}`, ` ${fixture.f1}`, `\t${fixture.f1}`, + `\`\`\`text\n${fixture.f1}\n\`\`\``, `~~~text\n${fixture.f1}\n~~~`]) expect(found(value)).toBe(false); + const first = withTrace('Read misses; write commits and deletes.'); + const last = withTrace('Reader then fills the pre-commit snapshot; every later reader sees stale data.').replace('| F1 |', '| F2 |'); + expect(found(first + '\n' + last)).toBe(false); + const cells = fixture.f1.split('|'); + cells[4] = ' Read misses; write commits and deletes. '; + cells[6] = ' Reader then fills the pre-commit snapshot; every later reader sees stale data. '; + expect(found(cells.join('|'))).toBe(false); + expect(found(first + '\n\n> ' + trace)).toBe(false); + expect(found(withTrace(`"${trace}" is a copied source example, not an observed defect.`))).toBe(false); +}); + +test('a real trace still rejects dismissal or acceptance of the later stale consequence', () => { + for (const suffix of [' No fix is required.', ' This stale-read behavior is accepted.', ' There is no stale-fill race.', + ' Later readers may return stale data and that is permitted.']) expect(found(withTrace(trace + suffix))).toBe(false); + expect(found(withTrace(trace + ' Original reader returns v1 to its own caller (allowed: it began before commit).'))).toBe(true); + expect(found(withTrace(trace + ' Later reader returns v1 to its own caller (allowed: it began after commit).'))).toBe(false); +}); +}); + +describe('sdk-original-order-ai', () => { +const captured = captured_sdk_original_order_ai; +const compact = () => `### Findings registry\n\n${captured.finding}\n\n${captured.heading}\n\`\`\`\n${captured.trace}\n\`\`\``; +const rejects = (changes: Array<[string, string]>) => { + for (const [before, after] of changes) { + expect(compact()).toContain(before); + expect(hasStaleFillRaceFinding(compact().replace(before, after))).toBe(false); + } +}; + +describe('asserted original order beside an amended cache schedule', () => { + test('exact completed report and its owned finding/schedule show the original late-fill violation', () => { + expect(hasStaleFillRaceFinding(captured.report)).toBe(true); + expect(hasStaleFillRaceFinding(compact())).toBe(true); + }); + + test('amended behavior or the original caller allowance cannot replace the original stale-fill evidence', () => { + const original = 'Original sketch, order A: fill V1 at 6 after delete at 4 -> R2 reads V1 for <=30 s VIOLATION'; + rejects([ + [original, ''], [original, 'Not ' + original], [original, '> ' + original], + [original, '"' + original + '"'], [original, 'If ' + original], + [original, original.replace('VIOLATION', 'PERMITTED')], + [original, original.replace('R2 reads', 'R1 reads')], + [original, original.replace('fill V1', 'skip fill V1')], + [original, original.replace('after delete at 4', 'before delete at 4')], + ]); + }); + + test('reader, writer, cache key, versions and completion order must all refer to the same execution', () => { + rejects([ + ['inflight[k]', 'inflight[foreign]'], ['R1 (began before W)', 'R1 (began after W)'], + ['DB write commits V2', 'DB write commits V1'], ['DB returns V1', 'DB returns V2'], + ['resume: invalidate(E1), delete', 'resume: invalidate(E9), delete'], + ['settles -> W complete', 'settles -> W pending'], + ['resume: E1.stale -> skip fill', 'resume: E9.stale -> skip fill'], + ['7 | R2 begins:', '4.5 | R2 begins:'], ['R2 reads V1 for', 'R2 reads V2 for'], + ['fill V1 at 6 after delete at 4', 'fill V1 at 3 after delete at 4'], + ['cache[k]', 'cache[foreign]'], + ]); + }); + + test('the current finding owns the trace and must independently assert the invariant violation', () => { + rejects([ + ['schedule (F1,', 'schedule (F2,'], ['| F1 | CRITICAL |', '| F2 | CRITICAL |'], + ['| F1 | CRITICAL |', '| F1 | LOW |'], + ['Schedule in Section 4 shows', 'A hypothetical Schedule in Section 4 shows'], + ['filled after `cache.delete`', 'filled before `cache.delete`'], + ['every read begun after that write completes must observe the committed version', 'earlier values are accepted for later readers'], + ]); + expect(hasStaleFillRaceFinding(compact().replace(captured.finding, captured.finding + '\n' + captured.finding))).toBe(false); + }); + + test('source and hypothetical framing cannot supply the assertion', () => { + for (const prefix of ['An unproven hypothesis.', 'Historical example only.', 'The following is a hypothetical example.']) { + expect(hasStaleFillRaceFinding(prefix + '\n' + compact())).toBe(false); + expect(hasStaleFillRaceFinding(compact().replace(captured.heading, prefix + '\n' + captured.heading))).toBe(false); + } + expect(hasStaleFillRaceFinding(compact().split('\n').map(line => '> ' + line).join('\n'))).toBe(false); + expect(hasStaleFillRaceFinding('````text\n' + compact() + '\n````')).toBe(false); + expect(hasStaleFillRaceFinding(compact().replace('### Findings registry', '### Quoted source'))).toBe(false); + }); + + test('same-finding direct and quoted withdrawals remain authoritative inside or after the trace', () => { + for (const withdrawal of ['F1 is withdrawn.', 'F1 is rejected.', 'The original schedule is impossible.', 'There is no stale-fill race.', 'Rejected: "There is no stale-fill race."']) { + expect(hasStaleFillRaceFinding(compact() + '\n\n' + withdrawal)).toBe(false); + expect(hasStaleFillRaceFinding(compact().replace(captured.trace, captured.trace + '\n' + withdrawal))).toBe(false); + expect(hasStaleFillRaceFinding(compact().replace('Ordering tests, both orders + late joiner + sentinel variant', withdrawal))).toBe(false); + } + expect(hasStaleFillRaceFinding(compact().replace('Readers that began before the write may still see the old snapshot (permitted by contract)', 'Later readers may see old snapshots; this stale-fill behavior is accepted.'))).toBe(false); + }); + + test('unrelated sections and consistently renamed identities do not change valid evidence', () => { + expect(hasStaleFillRaceFinding('### Prior example\nHistorical example only.\n\n### Current review\n' + compact())).toBe(true); + expect(hasStaleFillRaceFinding(compact() + '\n\n### Other finding\nF2 is rejected.')).toBe(true); + const renamed = compact().replaceAll('R1', 'R7').replaceAll('R2', 'R8').replaceAll('R3', 'R9') + .replaceAll('V1', 'oldSnapshot').replaceAll('V2', 'newSnapshot').replaceAll('E1', 'pendingA').replaceAll('E2', 'pendingB') + .replaceAll('[k]', '[profileKey]').replace(/\bW\b/g, 'W2'); + expect(hasStaleFillRaceFinding(renamed)).toBe(true); + }); + + test('owning source headings and same-finding assessments survive intervening structure', () => { + for (const heading of ['## Hypothetical example', '## Quoted source', '## Historical example only']) { + expect(hasStaleFillRaceFinding(heading + '\n' + compact())).toBe(false); + } + expect(hasStaleFillRaceFinding(compact().replace(captured.heading, + 'F1 is rejected.\n\nUnrelated diagram:\n```\nA -> B\n```\n\n' + captured.heading))).toBe(false); + expect(hasStaleFillRaceFinding(compact() + '\n\n### Assessment of F1\nF1 is rejected.')).toBe(false); + }); +}); + +const retry = () => `## Findings Registry\n\n${captured.retry.finding}\n\n${captured.retry.heading}\n\`\`\`\n${captured.retry.trace}\n\`\`\``; +describe('version-labelled original prose with its owned schedule', () => { + test('the exact retry and compact evidence require the original sequence, not amended prevention', () => { + expect(hasStaleFillRaceFinding(captured.retry.report)).toBe(true); + expect(hasStaleFillRaceFinding(retry())).toBe(true); + }); + + test('each version and shared key must agree, with write completion before the later reader', () => { + for (const [before, after] of [ + ['DB returns v1', 'DB returns v2'], ['write commits v2 and', 'write commits v1 and'], + ['read then fills v1;', 'read then fills v2;'], ['every later read gets v1', 'every later read gets v2'], + ['write commits v2 and', 'write commits v3 and'], ['writeGen[key]', 'writeGen[foreign]'], + ['cache[key]', 'cache[foreign]'], ['R2 (read, began after W)', 'R2 (read, began before W)'], + ['delete (no-op), return', 'delete (no-op), pending'], ['DB SELECT -> v1', 'DB SELECT -> v2'], + ['DB UPDATE commits v2', 'DB UPDATE commits v3'], ['promise resolves, set(v1)', 'promise resolves, set(v2)'], + ['get -> v1 VIOLATION', 'get -> v2 VIOLATION'], ['6 sketch', '3 sketch'], + ['3 DB UPDATE commits v2', '3 DB UPDATE commits v2'], + ]) { + expect(retry()).toContain(before); + expect(hasStaleFillRaceFinding(retry().replace(before, after))).toBe(false); + } + expect(hasStaleFillRaceFinding(retry().replace(captured.retry.trace, captured.retry.trace.split('\n').filter(line => !/\b[456] sketch\b/.test(line)).join('\n')))).toBe(false); + }); + + test('conditional, quoted, obsolete or withdrawn evidence cannot become a current finding', () => { + for (const prefix of ['An unproven hypothesis.', 'Historical example only.', 'The following is a hypothetical example.']) { + expect(hasStaleFillRaceFinding(prefix + '\n' + retry())).toBe(false); + expect(hasStaleFillRaceFinding(retry().replace('Late fill after write.', prefix + ' Late fill after write.'))).toBe(false); + } + for (const heading of ['## Hypothetical example', '## Quoted source', '## Historical example only']) { + expect(hasStaleFillRaceFinding(heading + '\n' + retry().replace('## Findings Registry', '### Findings Registry'))).toBe(false); + } + for (const withdrawal of ['F1 is rejected.', 'S1 is withdrawn.', 'The original schedule is impossible.', 'There is no stale-fill race.', 'Rejected: "There is no stale-fill race."']) { + expect(hasStaleFillRaceFinding(retry() + '\n\n' + withdrawal)).toBe(false); + expect(hasStaleFillRaceFinding(retry() + '\n\n### Assessment of F1\n' + withdrawal)).toBe(false); + expect(hasStaleFillRaceFinding(retry().replace(' S2 join stale flight', withdrawal + '\n S2 join stale flight'))).toBe(false); + } + for (const withdrawal of ['S1 is withdrawn.', 'F1 is rejected.']) { + expect(hasStaleFillRaceFinding(retry().replace(captured.retry.trace, captured.retry.trace + '\n' + withdrawal))).toBe(false); + } + expect(hasStaleFillRaceFinding(retry().split('\n').map(line => '> ' + line).join('\n'))).toBe(false); + expect(hasStaleFillRaceFinding('````\n' + retry() + '\n````')).toBe(false); + expect(hasStaleFillRaceFinding(retry().replace('Late fill after write.', 'If a late fill happens after write.'))).toBe(false); + }); + + test('consistent versions and independent later findings remain valid', () => { + expect(hasStaleFillRaceFinding(retry().replaceAll('v1', 'v7').replaceAll('v2', 'v8').replaceAll('[key]', '[profileKey]').replaceAll('key#1', 'profileKey#1'))).toBe(true); + expect(hasStaleFillRaceFinding('## Prior example\nHistorical only.\n\n## Current review\n' + retry().replace('## Findings Registry', '### Findings Registry'))).toBe(true); + expect(hasStaleFillRaceFinding(retry() + '\n\n### Other finding\nF9 is rejected.')).toBe(true); + }); +}); +}); + +describe('sdk-reported-coordination-ar', () => { +const fs = fs_sdk_ordered_schedule_ar; +const report = fs.readFileSync(new URL('./fixtures/sdk-reported-coordination-ar.md', import.meta.url), 'utf8'); +const paragraph = report.split('\n\n').find(text => text.startsWith('## Proposed wrapper integration'))!.split('\n').slice(1).join('\n'); +const matches = (text = paragraph) => hasStaleFillRaceFinding(text); + +test('the actual retry independently reports the original coordination violation', () => { + expect(paragraph).toContain('review found that this violates the read-after-write rule above (F1)'); + expect(paragraph).not.toMatch(/stale|in-flight|race|pending/); + expect(matches()).toBe(true); + expect(matches(report)).toBe(true); + expect(matches(paragraph.replace('proposed no coordination', 'had no coordination'))).toBe(true); + expect(matches(paragraph.replace('proposed no coordination', 'has no coordination'))).toBe(true); + expect(matches(paragraph.replace('sketch', 'wrapper'))).toBe(true); + expect(matches(paragraph.replace('rule above', 'contract'))).toBe(true); + expect(matches(paragraph.replace(/ and omits[\s\S]*/, '.'))).toBe(true); +}); + +test('missing or hypothetical premise and conclusion cannot become findings', () => { + for (const [from, to] of [ + ['proposed no coordination', 'proposed coordination'], + ['proposed no coordination', 'may propose no coordination'], + ['review found that this violates', 'review may find that this violates'], + ['review found that this violates', 'review found that this does not violate'], + ['review found that this violates', 'review hypothesized that this violates'], + ['review found that this violates', 'review found that another wrapper violates'], + ['read-after-write rule above', 'formatting rule'], + ['(F1)', '(unknown)'], + ['; the\nreview found', '. Another unrelated finding. The\nreview found'], + ['; the\nreview found', '\n\nThe\nreview found'], + ['; the\nreview found', ' | The\nreview found'], + ]) { + expect(paragraph).toContain(from); + expect(matches(paragraph.replace(from, to))).toBe(false); + } +}); + +test('source and quoted evidence cannot assert the current violation', () => { + for (const text of [ + 'Source:\n\n' + paragraph, + 'Hypothetical scenario. ' + paragraph, + 'Earlier review:\n\n' + paragraph, + '## Historical example\n' + paragraph, + '> ' + paragraph.replaceAll('\n', '\n> '), + '```text\n' + paragraph + '\n```', + '~~~text\n' + paragraph + '\n~~~', + paragraph.replace('original sketch proposed no coordination between a cache fill and a write', '`original sketch proposed no coordination between a cache fill and a write`'), + paragraph.replace('review found that this violates the read-after-write rule above (F1)', '"review found that this violates the read-after-write rule above (F1)"'), + ]) expect(matches(text)).toBe(false); +}); + +test('the referenced finding owns its later assessment', () => { + for (const tail of ['F1 is withdrawn.', 'F1 is "withdrawn".', 'F1 is rejected.', 'This finding is dismissed.', 'No coordination is required.', '| ID | Assessment |\n| F1 | Withdrawn: no coordination is required. |', '| F1 | Withdrawn |', '| F1 | "rejected" |']) { + expect(matches(paragraph + '\n\n' + tail)).toBe(false); + } + expect(matches(paragraph + '\n\nF2 is withdrawn.')).toBe(true); + expect(matches(paragraph + '\n\n| F2 | Withdrawn |')).toBe(true); + expect(matches(paragraph + '\n\n## Historical assessment\n| F1 | Withdrawn |')).toBe(true); + expect(matches('## Earlier material\nSource:\nOld source.\n\n## Current findings\n' + paragraph)).toBe(true); +}); +}); + +describe('sdk-schedule-continuation-ah', () => { +const fixture = fixture_sdk_schedule_continuation_ah; +const frame = fixture.compact; +function replace(from: string, to: string, input = frame): string { + expect(input.includes(from)).toBe(true); + return input.replace(from, to); +} +const originalRows = ' S2* | await read ... | write commits, delete(noop) | | - |\n' + + ' | resolves V0 → set V0 | | hit → V0 | V0 (30 s) | VIOLATION\n'; + +test('retains both exact public report forms as affirmative original-race findings', () => { + expect(hasStaleFillRaceFinding(fixture.report)).toBe(true); + expect(hasStaleFillRaceFinding(frame)).toBe(true); + expect(fixture.report.includes(frame.trim())).toBe(true); +}); + +test('binds consistently renamed actors, shared key, versions and finding/schedule IDs', () => { + const renamed = frame.replace(/\bR1\b/g, 'R7').replace(/\bR2\b/g, 'R8').replace(/\bW\b/g, 'W9') + .replace(/\bV0\b/g, 'oldValue').replace(/\bV1\b/g, 'freshValue') + .replace(/\bkey\b/g, 'profile_key').replace(/\bF1\b/g, 'F9').replace(/\bS2\b/g, 'S9'); + expect(hasStaleFillRaceFinding(renamed)).toBe(true); + expect(hasStaleFillRaceFinding(frame.replace(/→/g, '->'))).toBe(true); + const unrelated = '## Historical example\nAn unrelated old example.\n\n## Current findings\n\n'; + expect(hasStaleFillRaceFinding(unrelated + frame)).toBe(true); +}); + +test('amendments, permitted earlier readers and missing continuation do not supply the original race', () => { + for (const changed of [ + replace(originalRows, ''), + replace('S2* | await read', 'S2 A1 | await read'), + replace(originalRows, ' S2* | begins before W, joins | delete + forget | — | — | OK: R1 began before W completed (permitted clause)\n'), + replace('resolves V0 → set V0', 'resolves V0, slot gone→drop'), + replace('hit → V0', 'miss→read V1→set'), + replace('hit → V0', ''), + replace('VIOLATION\n S2 A1', 'OK (permitted earlier return)\n S2 A1'), + replace('VIOLATION\n S2 A1', 'VIOLATION\n | already guarded | | | | OK\n S2 A1'), + ]) expect(hasStaleFillRaceFinding(changed)).toBe(false); +}); + +test('requires the original schedule citation, legend and explicit post-completion boundary', () => { + for (const changed of [ + replace('Schedule S2 makes', 'Schedule S9 makes'), + replace('`*` = original sketch.', '`*` = amended sketch.'), + replace('`*` = original sketch.', ''), + replace('`*` = original sketch.', 'Hypothetically, `*` = original sketch.'), + replace('CRITICAL GAP | 1, 2, 4, 5, 6', 'CRITICAL GAP | 1, 2, 5, 6'), + replace('Violates retained invariant.', 'No defect in the retained invariant.'), + replace('R2 (begins after W)', 'R2 (begins before W)'), + replace('after `writeProfile` resolves', 'before `writeProfile` resolves'), + replace('after `writeProfile` resolves', 'after `writeProfile` begins'), + replace('after `writeProfile` resolves', 'after `readProfile` resolves'), + ]) expect(hasStaleFillRaceFinding(changed)).toBe(false); +}); + +test('rejects actor, key, value, invalidation and ordering mismatches', () => { + for (const changed of [ + replace('R2 (begins after W)', 'R1 (begins after W)'), + replace('R2 (begins after W)', 'R2 (begins after W9)'), + replace('`inflight[key]`', '`inflight[other_key]`'), + replace('| cache[key] | Result', '| cache[other_key] | Result'), + replace('W (commits V1)', 'W (commits V0)'), + replace('resolves V0 → set V0', 'resolves V1 → set V0'), + replace('resolves V0 → set V0', 'resolves V0 → set V1'), + replace('hit → V0', 'hit → V1'), + replace('write commits, delete(noop)', 'write begins, delete(noop)'), + replace('write commits, delete(noop)', 'write commits'), + replace('await read ...', 'await write ...'), + replace(originalRows, originalRows.split('\n').slice(0, 2).reverse().join('\n') + '\n'), + replace('V0 (30 s) | VIOLATION', 'V1 (30 s) | VIOLATION'), + replace('see V0 for 30 s;', 'see V1 for 30 s;'), + ]) expect(hasStaleFillRaceFinding(changed)).toBe(false); +}); + +test('quotes, source introductions and withdrawn findings remain negative', () => { + for (const prefix of ['An unproven hypothesis.', 'Historical example only.', 'The following is a hypothetical example.']) { + expect(hasStaleFillRaceFinding(prefix + '\n\n' + frame)).toBe(false); + expect(hasStaleFillRaceFinding(replace('### Async Ordering Record', prefix + '\n\n### Async Ordering Record'))).toBe(false); + expect(hasStaleFillRaceFinding(replace('### Findings Registry\n', '### Findings Registry\n\n' + prefix))).toBe(false); + } + expect(hasStaleFillRaceFinding(frame.split('\n').map(line => '> ' + line).join('\n'))).toBe(false); + expect(hasStaleFillRaceFinding('````text\n' + frame + '\n````')).toBe(false); + expect(hasStaleFillRaceFinding(replace('```\n Sched', '```javascript\n Sched'))).toBe(false); + for (const dismissal of [ + 'The original trace is impossible.', 'This schedule is not a bug.', + 'The original race is permitted.', 'The stale fill is accepted.', + 'No coordination is required.', + ]) { + expect(hasStaleFillRaceFinding(frame + '\n' + dismissal)).toBe(false); + expect(hasStaleFillRaceFinding(replace('Violates retained invariant.', 'Violates retained invariant. ' + dismissal))).toBe(false); + } +}); + + +test('completed prior decision section is independent; spoofed or withdrawn framing is not', () => { + const close = '### Decision Registry (all auto-resolved to recommended option)\n\n| D1 | A | B |\n\nLake Score: 7/7 recommendations chose the complete option.\n\n'; + expect(hasStaleFillRaceFinding(close + frame)).toBe(true); + expect(hasStaleFillRaceFinding(close.replace('### Decision Registry (all auto-resolved to recommended option)', '### Historical example') + frame)).toBe(false); + expect(hasStaleFillRaceFinding(close.replace('Lake Score: 7/7 recommendations chose the complete option.', 'An unproven hypothesis.') + frame)).toBe(false); + const row = frame.split('\n').find(line => line.startsWith('| F1 |'))!; + expect(hasStaleFillRaceFinding(replace(row, row + '\n' + row))).toBe(false); +}); +test('same finding or schedule tail withdrawals remain authoritative', () => { + for (const tail of ['S2 is impossible.', 'F1 is rejected. The original trace is impossible.', 'F1 is rejected.', 'S2 is withdrawn.']) { + expect(hasStaleFillRaceFinding(frame + '\n' + tail)).toBe(false); + } + expect(hasStaleFillRaceFinding(frame + '\nF2 is rejected. The original trace is impossible.')).toBe(true); + expect(hasStaleFillRaceFinding(frame + '\nS3 is impossible.')).toBe(true); +}); +}); + +describe('sdk-stale-table-ad-v3', () => { +const fixture = fixture_sdk_stale_table_ad_v3; +const found = hasStaleFillRaceFinding; +const allowance='Original reader still returns v1 to its own caller (allowed: it began before commit)'; +test('actual table finding distinguishes forbidden later stale reads from the permitted original caller',()=>{ + expect(found(fixture.report)).toBe(true); + expect(found(fixture.table)).toBe(true); + expect(fixture.provenance.noRetroactivePass).toBe(true); +}); +test('the already-started original read may use a version label without changing ownership',()=>{ + for(const token of ['v17','VERSION_A','snapshot-A'])expect(found(fixture.table.replaceAll('v1',token))).toBe(true); +}); +test('allowance cannot migrate to later readers, a post-commit start, or a cache fill',()=>{ + for(const changed of [ + 'Later readers return v1 (allowed: they began after commit)', + 'Original reader still returns v1 to its own caller (allowed: it began after commit)', + 'Original reader still returns v1 to its own caller (allowed: it never began before commit)', + 'Original reader fills the cache with v1 (allowed: it began before commit)', + 'Original reader still returns v1 to later readers (allowed: it began before commit)', + ])expect(found(fixture.table.replace(allowance,changed))).toBe(false); +}); +test('a permitted original caller cannot hide acceptance of later stale reads or no required fix',()=>{ + for(const suffix of [' This stale-read behavior is accepted.',' No fix is required.',' Later readers may return stale data; this is the accepted consistency model.']) + expect(found(fixture.table.replace('None against the invariant.','None against the invariant.'+suffix))).toBe(false); +}); +test('copied table source and absent late-fill evidence cannot provide coverage',()=>{ + expect(found('```text\n'+fixture.table+'\n```')).toBe(false); + expect(found(fixture.table.split('\n').map(x=>'> '+x).join('\n'))).toBe(false); + expect(found(fixture.table.split('\n').map(x=>' '+x).join('\n'))).toBe(false); + const rows=fixture.table.split('\n'),cells=rows[2]!.split('|'); + cells[4]=' There is no stale-fill race; later reads observe the committed value. '; + rows[2]=cells.join('|');expect(found(rows.join('\n'))).toBe(false); +}); + + +test('original-caller exception requires asserted chronology for that reader',()=>{ + for(const changed of [ + 'Original reader still returns v1 to its own caller (allowed: it may have begun before commit)', + 'Original reader still returns v1 to its own caller (allowed: it did not begin before commit)', + 'Original reader still returns v1 to its own caller (allowed: it began before commit only if the write failed)', + 'Original reader still returns v1 to its own caller (allowed: another reader began before commit)', + 'Original reader still returns v1 to its own caller (allowed: the write began before commit)', + 'If the original reader still returns v1 to its own caller, that is allowed: it began before commit', + ])expect(found(fixture.table.replace(allowance,changed))).toBe(false); +}); + +test('an original-return allowance cannot erase another allowed stale consequence',()=>{ + for(const changed of [ + allowance+' and stores that v1 in the cache for later readers', + allowance+'; later readers may reuse this old value and that is allowed', + allowance+'. New readers may reuse this old value and that is permitted', + allowance+'. The stale cache refill is acceptable', + ])expect(found(fixture.table.replace(allowance,changed))).toBe(false); +}); + +test('table rows cannot borrow an ordering defect from another issue or from quoted source',()=>{ + const rows=fixture.table.split('\n'),cells=rows[2]!.split('|'); + const originalFailure=cells[4]!; + cells[4]=' The original reader receives its pre-commit snapshot; later reads observe the committed version. '; + const missing=rows.slice(0,2).concat(cells.join('|')).join('\n'); + expect(found(missing)).toBe(false); + const other=cells.slice();other[1]=' D2 ';other[4]=originalFailure; + other[5]=' This stale-read behavior is accepted; no fix is required. '; + expect(found(missing+'\n'+other.join('|'))).toBe(false); + expect(found('> '+originalFailure+'\n\n'+missing)).toBe(false); + expect(found('```text\n'+originalFailure+'\n```\n\n'+missing)).toBe(false); +}); +}); diff --git a/test/coverage-audit-af.test.ts b/test/coverage-audit-af.test.ts deleted file mode 100644 index 37b8d915b..000000000 --- a/test/coverage-audit-af.test.ts +++ /dev/null @@ -1,147 +0,0 @@ -import { expect, test } from 'bun:test'; -import { coverageAuditVerdict } from './helpers/coverage-audit-evidence'; -import fixture from './fixtures/coverage-audit-af.json'; -import { E2E_TOUCHFILES } from './helpers/touchfiles'; -import { posix, win32 } from 'node:path'; -import { coverageAuditReadEvidence } from './helpers/coverage-audit-evidence'; - -const actual = (index: number) => structuredClone(fixture.rows[index]!); -const files = (row: typeof fixture.rows[number]) => ({cwd:row.cwd, - source:{path:`${row.cwd}/src/billing.ts`,content:fixture.files.source}, - tests:{path:`${row.cwd}/test/billing.test.ts`,content:fixture.files.tests}}); -for (let i=0;i { - const row=actual(i); expect(coverageAuditVerdict(row.result, files(row))).toEqual({sourceRead:true,testsRead:true,diagram:true,passed:true,failures:[]}); -}); - -function delivered(command: string, mutate?: (events: any[]) => void) { - const row=actual(2), session=row.sessionId; - const transcript:any[]=[ - {type:'system',subtype:'init',session_id:session,cwd:row.cwd}, - {type:'assistant',session_id:session,parent_tool_use_id:null,message:{role:'assistant',content:[{type:'tool_use',id:'read-pair',name:'Bash',input:{command}}]}}, - {type:'user',session_id:session,parent_tool_use_id:null,message:{role:'user',content:[{type:'tool_result',tool_use_id:'read-pair',is_error:false,content:fixture.files.source+'\n----\n'+fixture.files.tests}]}}, - ]; - mutate?.(transcript); - return coverageAuditVerdict({...row.result,transcript},files(row)); -} -const both = 'cat -n src/billing.ts && cat -n test/billing.test.ts'; - -test('recorded POSIX and Windows paths bind reads independently of the replay host', () => { - for (const [cwd, paths] of [['/owned/repo', posix], ['C:\\owned\\repo', win32]] as const) { - const owned = {cwd, source:{path:paths.join(cwd,'src/billing.ts'),content:fixture.files.source}, - tests:{path:paths.join(cwd,'test/billing.test.ts'),content:fixture.files.tests}}; - const transcript = [ - {type:'system',subtype:'init',session_id:'owned',cwd}, - {type:'assistant',session_id:'owned',message:{role:'assistant',content:[{type:'tool_use',id:'pair',name:'Bash',input:{command:both}}]}}, - {type:'user',session_id:'owned',message:{role:'user',content:[{type:'tool_result',tool_use_id:'pair',is_error:false,content:fixture.files.source+'\n'+fixture.files.tests}]}}, - ]; - expect(coverageAuditReadEvidence(transcript,owned)).toEqual({sourceRead:true,testsRead:true}); - expect(coverageAuditReadEvidence(transcript,{...owned,source:{...owned.source,path:paths.join(cwd,'../foreign.ts')}})) - .toEqual({sourceRead:false,testsRead:false}); - expect(coverageAuditReadEvidence(transcript,{...owned,source:{...owned.source,path:cwd+paths.sep+'src'+paths.sep+'..'+paths.sep+'src'+paths.sep+'billing.ts'}})) - .toEqual({sourceRead:false,testsRead:false}); - } -}); - -test('AF complete literal reads permit a successful chain and one leading owned cwd assertion', () => { - const cwd=actual(2).cwd; - for (const command of [both, `cd ${cwd}; cat -n src/billing.ts; cat -n test/billing.test.ts`, `cd '${cwd}' && ${both}`, 'cat -n src/billing.ts; echo ----; cat -n test/billing.test.ts']) { - const v=delivered(command); expect(v.sourceRead).toBe(true); expect(v.testsRead).toBe(true); expect(v.passed).toBe(true); - } -}); - -test('AF the new conditional-chain grammar conservatively rejects mixed separators', () => { - const v=delivered(`cd ${actual(2).cwd}; ${both}`); - expect(v.sourceRead).toBe(false); expect(v.testsRead).toBe(false); -}); - -test('AF read recognition rejects foreign or midstream cwd changes and nonliteral targets', () => { - const cwd=actual(2).cwd; - for (const command of [`cd /foreign; ${both}`, `cat -n src/billing.ts; cd /foreign; cat -n test/billing.test.ts`, - `cat -n src/billing.ts; cd ${cwd}; cat -n test/billing.test.ts`, `cd "$PWD"; ${both}`, `cd ${cwd}/..; ${both}`]) { - const v=delivered(command); expect(v.sourceRead).toBe(false); expect(v.testsRead).toBe(false); - } -}); - -test('AF a printed, conditional or skipped read cannot borrow delivered-looking file contents', () => { - for (const command of [`false && ${both}`, `if false; then ${both}; fi`, `echo '${both}'`, `exit; ${both}`, - `# ${both}`, `cat <<'EOF'\n${both}\nEOF`, `(${both})`, `f() { ${both}; }`, `printf '%s' '${both}'`, - `printf expected; false && ${both}; true`, `${both} > result.txt`]) { - const v=delivered(command); expect(v.sourceRead).toBe(false); expect(v.testsRead).toBe(false); - } -}); - -test('AF an owned cwd does not authorize mutations or interpreters around a read', () => { - for (const neighbor of ['rm -f src/billing.ts', 'python3 -c "pass"', 'echo fake > src/billing.ts', - 'grep data backup.txt | tee src/billing.ts', 'git diff --output=src/billing.ts', - "git diff '--output=src/billing.ts'", "git diff --output'='src/billing.ts", - 'git diff --out=src/billing.ts', 'git diff --ext-diff']) { - const result = delivered(`cd ${actual(2).cwd}; ${neighbor}; cat src/billing.ts; cat test/billing.test.ts`); - expect(result.sourceRead).toBe(false); expect(result.testsRead).toBe(false); - } -}); - -test('AF added command forms retain exact parent request/result success and delivered-content binding', () => { - const mutations:Array<(events:any[])=>void>=[ - e=>{e[0].cwd='/foreign';}, e=>{e[2].session_id='foreign';}, - e=>{e[1].parent_tool_use_id='child';}, e=>{e[2].message.content[0].is_error=true;}, - e=>{e[2].message.content[0].tool_use_id='unpaired';}, - e=>{e[2].message.content[0].content='The two filenames were read.';}, - ]; - for (const mutate of mutations) { const v=delivered(both,mutate); expect(v.sourceRead).toBe(false); expect(v.testsRead).toBe(false); } - const onlySource=delivered(both,e=>{e[2].message.content[0].content=fixture.files.source;}); - expect(onlySource.sourceRead).toBe(true); expect(onlySource.testsRead).toBe(false); -}); - -const flat = (legend = 'Legend: [✓] tested [✗] GAP') => `\`\`\`text\n${legend}\nprocessPayment(amount, currency)\n├── [✓] happy path USD\nrefundPayment(paymentId, reason)\n└── [✗] happy path refund\n\`\`\``; -function diagram(output:string) { const row=actual(0);return coverageAuditVerdict({...row.result,output},files(row)).diagram; } - -test('AF flat function roots and same-block legend symbols preserve seeded coverage ownership', () => { - expect(diagram(flat())).toBe(true); - expect(diagram(flat().replaceAll('✓','✔').replaceAll('✗','✘'))).toBe(true); - expect(diagram(flat().replace('processPayment(amount, currency)\n├── [✓] happy path USD\nrefundPayment(paymentId, reason)\n└── [✗] happy path refund', - 'refundPayment(paymentId, reason)\n├── [✗] happy path refund\nprocessPayment(amount, currency)\n└── [✓] happy path USD'))).toBe(true); -}); - -test('AF symbol-only markers need an unambiguous legend in their own diagram block', () => { - for (const output of [flat(''),flat('Legend: [✓] GAP [✗] tested'),flat('Legend: [✓] tested [✗] tested'), - flat('Legend: [✓] tested [✗] GAP [✗] tested'), - `\`\`\`text\nLegend: [✓] tested [✗] GAP\n\`\`\`\n${flat('')}`]) expect(diagram(output)).toBe(false); -}); - -test('AF flat roots cannot borrow another function subtree or a quoted/example diagram', () => { - for (const output of [ - flat().replace('├── [✓] happy path USD','├── untested amount guard\nunrelatedHelper()\n└── [✓] happy path USD'), - flat().replace('refundPayment(paymentId, reason)','unrelatedRefund(paymentId, reason)'), - flat().replace('processPayment(amount, currency)','processPaymentOther(amount, currency)'), - flat().split('\n').map(line=>'> '+line).join('\n'), - '````markdown\n'+flat()+'\n````', - flat().replace('Legend:','Example diagram:\nLegend:'), - ]) expect(diagram(output)).toBe(false); -}); - -test('AF a legend cannot override an explicitly negated marker on its own branch', () => { - for (const output of [ - flat().replace('[✗] happy path refund','not [✗] happy path refund'), - flat().replace('[✗] happy path refund','[✗] is false; this branch is tested'), - flat().replace('[✓] happy path USD','not [✓] happy path USD'), - ]) expect(diagram(output)).toBe(false); -}); - -test('AF coverage fixtures and controls select only the two existing coverage-audit owners', () => { - for (const file of ['test/coverage-audit-af.test.ts','test/fixtures/coverage-audit-af.json']) { - expect(Object.entries(E2E_TOUCHFILES).filter(([,paths])=>paths.includes(file)).map(([name])=>name).sort()).toEqual(['plan-eng-coverage-audit','review-coverage-audit']); - } -}); - - -test('AF symbol gaps retain affirmative legend ownership and reject same-branch contradictions', () => { - for (const output of [ - flat().replace('[✗] happy path refund', '[✗] happy path refund (marker is incorrect; this branch is fully tested)'), - flat().replace('[✗] happy path refund', '[✗] happy path refund — no coverage gap exists'), - flat('An unproven hypothesis: [✓] tested [✗] GAP'), - flat("The source says '[✓] tested [✗] GAP'"), - ]) expect(diagram(output)).toBe(false); - expect(diagram(flat())).toBe(true); - expect(diagram(flat('[✓] tested [✗] GAP'))).toBe(true); - expect(diagram(flat('src/billing.ts — test coverage map [✓] tested [✗] GAP'))).toBe(true); -}); diff --git a/test/coverage-audit-aw.test.ts b/test/coverage-audit-aw.test.ts deleted file mode 100644 index 0c70c042b..000000000 --- a/test/coverage-audit-aw.test.ts +++ /dev/null @@ -1,32 +0,0 @@ -import {describe,expect,test} from 'bun:test'; -import {coverageAuditReadEvidence,coverageAuditVerdict} from './helpers/coverage-audit-evidence'; -import fixture from './fixtures/coverage-audit-aw.json'; -const fresh=(n=0)=>{ - const r=structuredClone(fixture.reads[n]!),files={cwd:r.cwd,source:{path:r.cwd+'/src/billing.ts',content:fixture.source},tests:{path:r.cwd+'/test/billing.test.ts',content:fixture.tests}}; - const transcript:any[]=[{type:'system',subtype:'init',session_id:r.sessionId,cwd:r.cwd},{type:'assistant',session_id:r.sessionId,parent_tool_use_id:null,message:{role:'assistant',content:[{type:'tool_use',id:r.toolUseId,name:'Bash',input:{command:r.command}}]}},{type:'user',session_id:r.sessionId,parent_tool_use_id:null,message:{role:'user',content:[{type:'tool_result',tool_use_id:r.toolUseId,is_error:false,content:r.outputExcerpt}]}}]; - return {files,transcript}; -}; -const reads=(x:ReturnType)=>coverageAuditReadEvidence(x.transcript,x.files); -const diagram=(text:string)=>{const x=fresh();return coverageAuditVerdict({exitReason:'success',browseErrors:[],output:text,transcript:x.transcript} as any,x.files).diagram;}; -describe('Coverage audit owned display composition and marker continuations',()=>{ - test.each([0,1,2,3])('credits exact complete public file delivery %i',n=>{ - expect(fixture.provenance.actualCollectorFailuresRetained).toBe(true);expect(reads(fresh(n))).toEqual({sourceRead:true,testsRead:true}); - }); - test.each(['failed result','foreign session','foreign cwd','foreign tool id','sidechain','missing result','repeated result','partial body','forged body'])('rejects %s',form=>{ - for(let n=0;n<4;n++){const x=fresh(n),e=x.transcript[2],b=e.message.content[0];if(form==='failed result')b.is_error=true;else if(form==='foreign session')e.session_id='foreign';else if(form==='foreign cwd')x.transcript[0].cwd+='/other';else if(form==='foreign tool id')b.tool_use_id='foreign';else if(form==='sidechain')e.parent_tool_use_id='parent';else if(form==='missing result')x.transcript.pop();else if(form==='repeated result')x.transcript.push(structuredClone(e));else if(form==='partial body')b.content=b.content.replace(/.*(?:export function processPayment|import \{ describe).*\n/g,'');else b.content='src/billing.ts and test/billing.test.ts were read';expect(reads(x)).toEqual({sourceRead:false,testsRead:false});} - }); - test.each(['foreign paths','printf forgery','echo escape forgery','expansion','double quoted expansion','awk execution','changed ordered prefix'])('rejects unsupported or unowned command: %s',form=>{ - const n=form==='awk execution'?2:form==='changed ordered prefix'?3:0,x=fresh(n),u=x.transcript[1].message.content[0];u.input.command=form==='foreign paths'?u.input.command.replaceAll('src/billing.ts','other/billing.ts').replaceAll('test/billing.test.ts','other/billing.test.ts'):form==='printf forgery'?"printf 'fixture body'":form==='echo escape forgery'?"echo -e 'fake\\nbody'":form==='expansion'?u.input.command+'; echo $(cat source)':form==='double quoted expansion'?u.input.command+'; echo "$HOME"':form==='awk execution'?u.input.command.replace('{f=1}','{system("cat forged") }'):u.input.command.replace('=== src/billing.ts ===','=== other.ts ===');expect(reads(x)).toEqual({sourceRead:false,testsRead:false}); - }); - test.each([0,1])('accepts the exact public current diagram %i',n=>expect(diagram(fixture.diagrams[n]!.text)).toBe(true)); - test.each(['missing key','inverted checkbox key','withdrawn key','foreign function','quoted source','not covered','not missing'])('rejects contradictory or unowned checkbox coverage: %s',form=>{ - const text=fixture.diagrams[0]!.text;const changed=form==='missing key'?text.replace(/^Legend:.*\n/m,''):form==='inverted checkbox key'?text.replace('[x] tested [ ] GAP','[x] untested [ ] tested'):form==='withdrawn key'?text.replace('src/billing.ts\n│','This legend is withdrawn.\nsrc/billing.ts\n│'):form==='foreign function'?text.replaceAll('refundPayment','otherPayment'):form==='quoted source'?'Example only:\n'+text:form==='not covered'?text.replace("[x] 'processes valid payment'","[ ] GAP"):text.replaceAll('[ ] GAP','[x] tested');expect(diagram(changed)).toBe(false); - }); - test('continuations keep their own row and cannot borrow from prose or a distant column',()=>{ - const text=fixture.diagrams[1]!.text; - expect(diagram(text.replace('│ [✓] billing.test.ts:6', '│ Earlier example:\n│ [✓] billing.test.ts:6'))).toBe(false); - expect(diagram(text.replace('│ [✓] billing.test.ts:6', ' [✓] billing.test.ts:6'))).toBe(false); - expect(diagram(text.replace('│ [✓] billing.test.ts:6', '│ [✗] billing.test.ts:6'))).toBe(false); - expect(diagram(text.replaceAll('[✗] GAP','[✓] tested').replace('[✗] untested (GAP)','[✗] untested (GAP)'))).toBe(false); - }); -}); diff --git a/test/coverage-audit-evidence.test.ts b/test/coverage-audit-evidence.test.ts index a27a7fd9a..7f57e7ec7 100644 --- a/test/coverage-audit-evidence.test.ts +++ b/test/coverage-audit-evidence.test.ts @@ -6,6 +6,16 @@ import { coverageAuditVerdict } from './helpers/coverage-audit-evidence'; import { recordE2E } from './helpers/e2e-helpers'; import { E2E_TOUCHFILES, LLM_JUDGE_TOUCHFILES, GLOBAL_TOUCHFILES } from './helpers/touchfiles'; import { selectTests } from './helpers/test-selection'; +import fixture_coverage_audit_af from './fixtures/coverage-audit-af.json'; +import { posix } from 'node:path'; +import { win32 } from 'node:path'; +import { coverageAuditReadEvidence } from './helpers/coverage-audit-evidence'; +import fixture_coverage_audit_aw from './fixtures/coverage-audit-aw.json'; +import captured_coverage_audit_shell_legend_at from './fixtures/coverage-audit-shell-legend-at.json'; +import fixture_coverage_checkbox_tail_av from './fixtures/coverage-checkbox-tail-av.json'; +import captured_coverage_diagram_legend_as from './fixtures/coverage-diagram-legend-as.json'; +import fixture_coverage_shell_display_aq from './fixtures/coverage-shell-display-aq.json'; +import billing_coverage_shell_display_aq from './fixtures/coverage-audit-ae.json'; const clone = (v:T):T => structuredClone(v); const diagram = '```text\nsrc/billing.ts\n├── processPayment: happy path [TESTED]\n└── refundPayment [UNTESTED]\n```'; @@ -442,8 +452,542 @@ Guard clauses tested: 0 / 4 }); test('coverage evidence files select their exact registered consumers',()=>{ for(const file of ['test/helpers/coverage-audit-evidence.ts','test/coverage-audit-evidence.test.ts','test/fixtures/coverage-audit-ae.json','test/fixtures/coverage-audit-ci-diagrams.json']){ - expect(selectTests([file],E2E_TOUCHFILES,GLOBAL_TOUCHFILES).selected.sort()).toEqual(file === 'test/helpers/coverage-audit-evidence.ts' ? ['plan-eng-coverage-audit','review-coverage-audit','ship-coverage-audit'] : ['plan-eng-coverage-audit','review-coverage-audit']); + expect(selectTests([file],E2E_TOUCHFILES,GLOBAL_TOUCHFILES).selected.sort()).toEqual(file === 'test/helpers/coverage-audit-evidence.ts' || file === 'test/coverage-audit-evidence.test.ts' ? ['plan-eng-coverage-audit','review-coverage-audit','ship-coverage-audit'] : ['plan-eng-coverage-audit','review-coverage-audit']); expect(selectTests([file],LLM_JUDGE_TOUCHFILES,GLOBAL_TOUCHFILES).selected).toEqual([]); } }); }); + +describe('coverage-audit-af', () => { +const fixture = fixture_coverage_audit_af; +const actual = (index: number) => structuredClone(fixture.rows[index]!); +const files = (row: typeof fixture.rows[number]) => ({cwd:row.cwd, + source:{path:`${row.cwd}/src/billing.ts`,content:fixture.files.source}, + tests:{path:`${row.cwd}/test/billing.test.ts`,content:fixture.files.tests}}); +for (let i=0;i { + const row=actual(i); expect(coverageAuditVerdict(row.result, files(row))).toEqual({sourceRead:true,testsRead:true,diagram:true,passed:true,failures:[]}); +}); + +function delivered(command: string, mutate?: (events: any[]) => void) { + const row=actual(2), session=row.sessionId; + const transcript:any[]=[ + {type:'system',subtype:'init',session_id:session,cwd:row.cwd}, + {type:'assistant',session_id:session,parent_tool_use_id:null,message:{role:'assistant',content:[{type:'tool_use',id:'read-pair',name:'Bash',input:{command}}]}}, + {type:'user',session_id:session,parent_tool_use_id:null,message:{role:'user',content:[{type:'tool_result',tool_use_id:'read-pair',is_error:false,content:fixture.files.source+'\n----\n'+fixture.files.tests}]}}, + ]; + mutate?.(transcript); + return coverageAuditVerdict({...row.result,transcript},files(row)); +} +const both = 'cat -n src/billing.ts && cat -n test/billing.test.ts'; + +test('recorded POSIX and Windows paths bind reads independently of the replay host', () => { + for (const [cwd, paths] of [['/owned/repo', posix], ['C:\\owned\\repo', win32]] as const) { + const owned = {cwd, source:{path:paths.join(cwd,'src/billing.ts'),content:fixture.files.source}, + tests:{path:paths.join(cwd,'test/billing.test.ts'),content:fixture.files.tests}}; + const transcript = [ + {type:'system',subtype:'init',session_id:'owned',cwd}, + {type:'assistant',session_id:'owned',message:{role:'assistant',content:[{type:'tool_use',id:'pair',name:'Bash',input:{command:both}}]}}, + {type:'user',session_id:'owned',message:{role:'user',content:[{type:'tool_result',tool_use_id:'pair',is_error:false,content:fixture.files.source+'\n'+fixture.files.tests}]}}, + ]; + expect(coverageAuditReadEvidence(transcript,owned)).toEqual({sourceRead:true,testsRead:true}); + expect(coverageAuditReadEvidence(transcript,{...owned,source:{...owned.source,path:paths.join(cwd,'../foreign.ts')}})) + .toEqual({sourceRead:false,testsRead:false}); + expect(coverageAuditReadEvidence(transcript,{...owned,source:{...owned.source,path:cwd+paths.sep+'src'+paths.sep+'..'+paths.sep+'src'+paths.sep+'billing.ts'}})) + .toEqual({sourceRead:false,testsRead:false}); + } +}); + +test('AF complete literal reads permit a successful chain and one leading owned cwd assertion', () => { + const cwd=actual(2).cwd; + for (const command of [both, `cd ${cwd}; cat -n src/billing.ts; cat -n test/billing.test.ts`, `cd '${cwd}' && ${both}`, 'cat -n src/billing.ts; echo ----; cat -n test/billing.test.ts']) { + const v=delivered(command); expect(v.sourceRead).toBe(true); expect(v.testsRead).toBe(true); expect(v.passed).toBe(true); + } +}); + +test('AF the new conditional-chain grammar conservatively rejects mixed separators', () => { + const v=delivered(`cd ${actual(2).cwd}; ${both}`); + expect(v.sourceRead).toBe(false); expect(v.testsRead).toBe(false); +}); + +test('AF read recognition rejects foreign or midstream cwd changes and nonliteral targets', () => { + const cwd=actual(2).cwd; + for (const command of [`cd /foreign; ${both}`, `cat -n src/billing.ts; cd /foreign; cat -n test/billing.test.ts`, + `cat -n src/billing.ts; cd ${cwd}; cat -n test/billing.test.ts`, `cd "$PWD"; ${both}`, `cd ${cwd}/..; ${both}`]) { + const v=delivered(command); expect(v.sourceRead).toBe(false); expect(v.testsRead).toBe(false); + } +}); + +test('AF a printed, conditional or skipped read cannot borrow delivered-looking file contents', () => { + for (const command of [`false && ${both}`, `if false; then ${both}; fi`, `echo '${both}'`, `exit; ${both}`, + `# ${both}`, `cat <<'EOF'\n${both}\nEOF`, `(${both})`, `f() { ${both}; }`, `printf '%s' '${both}'`, + `printf expected; false && ${both}; true`, `${both} > result.txt`]) { + const v=delivered(command); expect(v.sourceRead).toBe(false); expect(v.testsRead).toBe(false); + } +}); + +test('AF an owned cwd does not authorize mutations or interpreters around a read', () => { + for (const neighbor of ['rm -f src/billing.ts', 'python3 -c "pass"', 'echo fake > src/billing.ts', + 'grep data backup.txt | tee src/billing.ts', 'git diff --output=src/billing.ts', + "git diff '--output=src/billing.ts'", "git diff --output'='src/billing.ts", + 'git diff --out=src/billing.ts', 'git diff --ext-diff']) { + const result = delivered(`cd ${actual(2).cwd}; ${neighbor}; cat src/billing.ts; cat test/billing.test.ts`); + expect(result.sourceRead).toBe(false); expect(result.testsRead).toBe(false); + } +}); + +test('AF added command forms retain exact parent request/result success and delivered-content binding', () => { + const mutations:Array<(events:any[])=>void>=[ + e=>{e[0].cwd='/foreign';}, e=>{e[2].session_id='foreign';}, + e=>{e[1].parent_tool_use_id='child';}, e=>{e[2].message.content[0].is_error=true;}, + e=>{e[2].message.content[0].tool_use_id='unpaired';}, + e=>{e[2].message.content[0].content='The two filenames were read.';}, + ]; + for (const mutate of mutations) { const v=delivered(both,mutate); expect(v.sourceRead).toBe(false); expect(v.testsRead).toBe(false); } + const onlySource=delivered(both,e=>{e[2].message.content[0].content=fixture.files.source;}); + expect(onlySource.sourceRead).toBe(true); expect(onlySource.testsRead).toBe(false); +}); + +const flat = (legend = 'Legend: [✓] tested [✗] GAP') => `\`\`\`text\n${legend}\nprocessPayment(amount, currency)\n├── [✓] happy path USD\nrefundPayment(paymentId, reason)\n└── [✗] happy path refund\n\`\`\``; +function diagram(output:string) { const row=actual(0);return coverageAuditVerdict({...row.result,output},files(row)).diagram; } + +test('AF flat function roots and same-block legend symbols preserve seeded coverage ownership', () => { + expect(diagram(flat())).toBe(true); + expect(diagram(flat().replaceAll('✓','✔').replaceAll('✗','✘'))).toBe(true); + expect(diagram(flat().replace('processPayment(amount, currency)\n├── [✓] happy path USD\nrefundPayment(paymentId, reason)\n└── [✗] happy path refund', + 'refundPayment(paymentId, reason)\n├── [✗] happy path refund\nprocessPayment(amount, currency)\n└── [✓] happy path USD'))).toBe(true); +}); + +test('AF symbol-only markers need an unambiguous legend in their own diagram block', () => { + for (const output of [flat(''),flat('Legend: [✓] GAP [✗] tested'),flat('Legend: [✓] tested [✗] tested'), + flat('Legend: [✓] tested [✗] GAP [✗] tested'), + `\`\`\`text\nLegend: [✓] tested [✗] GAP\n\`\`\`\n${flat('')}`]) expect(diagram(output)).toBe(false); +}); + +test('AF flat roots cannot borrow another function subtree or a quoted/example diagram', () => { + for (const output of [ + flat().replace('├── [✓] happy path USD','├── untested amount guard\nunrelatedHelper()\n└── [✓] happy path USD'), + flat().replace('refundPayment(paymentId, reason)','unrelatedRefund(paymentId, reason)'), + flat().replace('processPayment(amount, currency)','processPaymentOther(amount, currency)'), + flat().split('\n').map(line=>'> '+line).join('\n'), + '````markdown\n'+flat()+'\n````', + flat().replace('Legend:','Example diagram:\nLegend:'), + ]) expect(diagram(output)).toBe(false); +}); + +test('AF a legend cannot override an explicitly negated marker on its own branch', () => { + for (const output of [ + flat().replace('[✗] happy path refund','not [✗] happy path refund'), + flat().replace('[✗] happy path refund','[✗] is false; this branch is tested'), + flat().replace('[✓] happy path USD','not [✓] happy path USD'), + ]) expect(diagram(output)).toBe(false); +}); +test('AF symbol gaps retain affirmative legend ownership and reject same-branch contradictions', () => { + for (const output of [ + flat().replace('[✗] happy path refund', '[✗] happy path refund (marker is incorrect; this branch is fully tested)'), + flat().replace('[✗] happy path refund', '[✗] happy path refund — no coverage gap exists'), + flat('An unproven hypothesis: [✓] tested [✗] GAP'), + flat("The source says '[✓] tested [✗] GAP'"), + ]) expect(diagram(output)).toBe(false); + expect(diagram(flat())).toBe(true); + expect(diagram(flat('[✓] tested [✗] GAP'))).toBe(true); + expect(diagram(flat('src/billing.ts — test coverage map [✓] tested [✗] GAP'))).toBe(true); +}); +}); + +describe('coverage-audit-aw', () => { +const fixture = fixture_coverage_audit_aw; +const fresh=(n=0)=>{ + const r=structuredClone(fixture.reads[n]!),files={cwd:r.cwd,source:{path:r.cwd+'/src/billing.ts',content:fixture.source},tests:{path:r.cwd+'/test/billing.test.ts',content:fixture.tests}}; + const transcript:any[]=[{type:'system',subtype:'init',session_id:r.sessionId,cwd:r.cwd},{type:'assistant',session_id:r.sessionId,parent_tool_use_id:null,message:{role:'assistant',content:[{type:'tool_use',id:r.toolUseId,name:'Bash',input:{command:r.command}}]}},{type:'user',session_id:r.sessionId,parent_tool_use_id:null,message:{role:'user',content:[{type:'tool_result',tool_use_id:r.toolUseId,is_error:false,content:r.outputExcerpt}]}}]; + return {files,transcript}; +}; +const reads=(x:ReturnType)=>coverageAuditReadEvidence(x.transcript,x.files); +const diagram=(text:string)=>{const x=fresh();return coverageAuditVerdict({exitReason:'success',browseErrors:[],output:text,transcript:x.transcript} as any,x.files).diagram;}; +describe('Coverage audit owned display composition and marker continuations',()=>{ + test.each([0,1,2,3])('credits exact complete public file delivery %i',n=>{ + expect(fixture.provenance.actualCollectorFailuresRetained).toBe(true);expect(reads(fresh(n))).toEqual({sourceRead:true,testsRead:true}); + }); + test.each(['failed result','foreign session','foreign cwd','foreign tool id','sidechain','missing result','repeated result','partial body','forged body'])('rejects %s',form=>{ + for(let n=0;n<4;n++){const x=fresh(n),e=x.transcript[2],b=e.message.content[0];if(form==='failed result')b.is_error=true;else if(form==='foreign session')e.session_id='foreign';else if(form==='foreign cwd')x.transcript[0].cwd+='/other';else if(form==='foreign tool id')b.tool_use_id='foreign';else if(form==='sidechain')e.parent_tool_use_id='parent';else if(form==='missing result')x.transcript.pop();else if(form==='repeated result')x.transcript.push(structuredClone(e));else if(form==='partial body')b.content=b.content.replace(/.*(?:export function processPayment|import \{ describe).*\n/g,'');else b.content='src/billing.ts and test/billing.test.ts were read';expect(reads(x)).toEqual({sourceRead:false,testsRead:false});} + }); + test.each(['foreign paths','printf forgery','echo escape forgery','expansion','double quoted expansion','awk execution','changed ordered prefix'])('rejects unsupported or unowned command: %s',form=>{ + const n=form==='awk execution'?2:form==='changed ordered prefix'?3:0,x=fresh(n),u=x.transcript[1].message.content[0];u.input.command=form==='foreign paths'?u.input.command.replaceAll('src/billing.ts','other/billing.ts').replaceAll('test/billing.test.ts','other/billing.test.ts'):form==='printf forgery'?"printf 'fixture body'":form==='echo escape forgery'?"echo -e 'fake\\nbody'":form==='expansion'?u.input.command+'; echo $(cat source)':form==='double quoted expansion'?u.input.command+'; echo "$HOME"':form==='awk execution'?u.input.command.replace('{f=1}','{system("cat forged") }'):u.input.command.replace('=== src/billing.ts ===','=== other.ts ===');expect(reads(x)).toEqual({sourceRead:false,testsRead:false}); + }); + test.each([0,1])('accepts the exact public current diagram %i',n=>expect(diagram(fixture.diagrams[n]!.text)).toBe(true)); + test.each(['missing key','inverted checkbox key','withdrawn key','foreign function','quoted source','not covered','not missing'])('rejects contradictory or unowned checkbox coverage: %s',form=>{ + const text=fixture.diagrams[0]!.text;const changed=form==='missing key'?text.replace(/^Legend:.*\n/m,''):form==='inverted checkbox key'?text.replace('[x] tested [ ] GAP','[x] untested [ ] tested'):form==='withdrawn key'?text.replace('src/billing.ts\n│','This legend is withdrawn.\nsrc/billing.ts\n│'):form==='foreign function'?text.replaceAll('refundPayment','otherPayment'):form==='quoted source'?'Example only:\n'+text:form==='not covered'?text.replace("[x] 'processes valid payment'","[ ] GAP"):text.replaceAll('[ ] GAP','[x] tested');expect(diagram(changed)).toBe(false); + }); + test('continuations keep their own row and cannot borrow from prose or a distant column',()=>{ + const text=fixture.diagrams[1]!.text; + expect(diagram(text.replace('│ [✓] billing.test.ts:6', '│ Earlier example:\n│ [✓] billing.test.ts:6'))).toBe(false); + expect(diagram(text.replace('│ [✓] billing.test.ts:6', ' [✓] billing.test.ts:6'))).toBe(false); + expect(diagram(text.replace('│ [✓] billing.test.ts:6', '│ [✗] billing.test.ts:6'))).toBe(false); + expect(diagram(text.replaceAll('[✗] GAP','[✓] tested').replace('[✗] untested (GAP)','[✗] untested (GAP)'))).toBe(false); + }); +}); +}); + +describe('coverage-audit-shell-legend-at', () => { +const captured = captured_coverage_audit_shell_legend_at; +const both = { sourceRead: true, testsRead: true }; +const neither = { sourceRead: false, testsRead: false }; +function owned(index: number) { + const row = structuredClone(captured[index]!) as any; + const useEvent = row.result.transcript.find((event: any) => event.message?.content.some((block: any) => + block.type === 'tool_use' && block.name === 'Bash' && block.input.command.includes('cat -n src/billing.ts'))); + const use = useEvent.message.content.find((block: any) => block.type === 'tool_use' && block.name === 'Bash' && block.input.command.includes('cat -n src/billing.ts')); + const resultEvent = row.result.transcript.find((event: any) => event.message?.content.some((block: any) => block.type === 'tool_result' && block.tool_use_id === use.id)); + row.result.transcript = [row.result.transcript.find((event: any) => event.type === 'system' && event.subtype === 'init'), useEvent, resultEvent]; + return { row, use, resultEvent, delivered: resultEvent.message.content.find((block: any) => block.tool_use_id === use.id) }; +} +function reads(index: number, mutate?: (s: ReturnType) => void) { + const s = owned(index); mutate?.(s); + return coverageAuditReadEvidence(s.row.result.transcript, s.row.files); +} +const flat = (legend = 'Legend [ OK ] covered [ GAP ] no test') => + '```text\nprocessPayment(amount, currency)\n├── happy path return success [ OK ]\nrefundPayment(paymentId, reason)\n└── happy path return refunded [ GAP ]\n' + legend + '\n```'; +const diagram = (output: string) => coverageAuditVerdict({ ...captured[1]!.result, output } as any, captured[1]!.files).diagram; + +test('exact public AT first, retry and engineering outputs retain all required native evidence', () => { + expect(captured.map(row => row.recordedPassed)).toEqual([false, false, true]); + for (const row of captured) expect(coverageAuditVerdict(row.result as any, row.files)).toEqual({ ...both, diagram: true, passed: true, failures: [] }); + expect(reads(0)).toEqual(both); expect(reads(1)).toEqual(both); +}); + +test('literal grep display options and filename captions do not own source bytes', () => { + for (const flags of ['-n', '-n -i', '-n -B1 -A200', '-n -i -B3 -A40']) { + expect(reads(0, s => { s.use.input.command = s.use.input.command.replace('-n -i -B3 -A40', flags); })).toEqual(both); + } + for (const replace of ['echo \'=== another-file.md ===\'', 'echo "--- src/billing.ts ---"', 'echo']) { + expect(reads(1, s => { s.use.input.command = s.use.input.command.replace('echo "=== testing.md ==="', replace); })).toEqual(both); + } + expect(reads(1, s => { s.use.input.command = s.use.input.command.replace('git diff main --stat', 'git diff HEAD~1 --stat'); })).toEqual(both); +}); + +test('escaped grep patterns keep a closed flag and literal operand grammar', () => { + for (const replacement of ['-n -i -B3 -A40 -f other', '-n -i --include=*', '-n -B-1', '-n -A100000', '-n -i -B3 -A40; false']) { + expect(reads(0, s => { s.use.input.command = s.use.input.command.replace('-n -i -B3 -A40', replacement); })).toEqual(neither); + } + for (const operand of ['-f/tmp/foreign', '"-f/tmp/foreign"', 'review/SKILL.md --include=*']) { + expect(reads(0, s => { s.use.input.command = s.use.input.command.replace('review/SKILL.md |', operand + ' |'); })).toEqual(neither); + } +}); + +test('successful conditional display paths reject execution, substitutions and hidden failure', () => { + for (const replacement of [ + 'echo -e "=== testing.md ==="', 'printf "=== testing.md ==="', 'echo "$(cat fake)"', 'echo `cat fake`', + 'echo "=== testing.md ==="; false', 'false || echo "=== testing.md ==="', 'unknown', + 'echo "cat -n src/billing.ts"', 'echo "=== testing.md ===\\nreplacement"', + 'cd ../sibling', 'env PATH=/tmp cat fake', 'echo "=== testing.md ===" > src/billing.ts', + ]) expect(reads(1, s => { s.use.input.command = s.use.input.command.replace('echo "=== testing.md ==="', replacement); })).toEqual(neither); + for (const command of ['git diff --ext-diff --stat', 'git diff main --output=src/billing.ts --stat', 'git -c core.pager=evil diff main --stat', 'git diff --no-index main --stat', 'git diff main --stat || echo ok']) { + expect(reads(1, s => { s.use.input.command = s.use.input.command.replace('git diff main --stat', command); })).toEqual(neither); + } +}); + +test('a valid display path still requires one complete successful owned delivery', () => { + for (const index of [0, 1]) for (const mutate of [ + (s: ReturnType) => { s.delivered.is_error = true; }, + (s: ReturnType) => { s.delivered.content = 'src/billing.ts and test/billing.test.ts were read'; }, + (s: ReturnType) => { s.delivered.content = s.row.files.source.content.slice(0, 80); }, + (s: ReturnType) => { s.resultEvent.session_id = 'foreign'; }, + (s: ReturnType) => { s.resultEvent.parent_tool_use_id = 'child'; }, + (s: ReturnType) => { s.delivered.tool_use_id = 'foreign'; }, + (s: ReturnType) => { s.row.result.transcript.push(structuredClone(s.resultEvent)); }, + (s: ReturnType) => { s.use.input.command = s.use.input.command.replace('cat -n src/billing.ts', 'echo src/billing.ts').replace('cat -n test/billing.test.ts', 'echo test/billing.test.ts'); }, + ]) expect(reads(index, mutate)).toEqual(neither); + expect(reads(1, s => { s.delivered.content = s.row.files.source.content; })).toEqual({ sourceRead: true, testsRead: false }); + expect(reads(1, s => { s.delivered.content = s.row.files.tests.content; })).toEqual({ sourceRead: false, testsRead: true }); +}); + +test('text statuses use the declared local meanings with whitespace and either pair order', () => { + for (const legend of ['Legend [ OK ] covered [ GAP ] no test', 'Legend: [OK] tested | [GAP] untested', 'Legend: [ GAP ] no test; [ OK ] covered']) expect(diagram(flat(legend))).toBe(true); + expect(diagram(flat().replaceAll('[ OK ]', '[OK]').replaceAll('[ GAP ]', '[GAP]'))).toBe(true); + expect(diagram(flat().replace('Legend [ OK ] covered [ GAP ] no test\n', '').replace('processPayment', 'Legend [ OK ] covered [ GAP ] no test\nprocessPayment'))).toBe(true); +}); + +test('missing, malformed, contradictory or foreign text legends cannot grant coverage', () => { + for (const legend of ['', '> Legend [ OK ] covered [ GAP ] no test', '"Legend [ OK ] covered [ GAP ] no test"', + 'Example: Legend [ OK ] covered [ GAP ] no test', 'If enabled, Legend [ OK ] covered [ GAP ] no test', + 'Legend [ OK ] no test [ GAP ] covered', 'Legend [ OK ] covered [ GAP ] covered', + 'Legend [ OK ] covered [ OK ] no test', 'Legend [ OK ] covered [ GAP ] no test except refunds', + 'Legend [ OK ] covered [ GAP ] no test\nLegend [ OK ] no test [ GAP ] covered', + ]) expect(diagram(flat(legend))).toBe(false); + expect(diagram('```text\nLegend [ OK ] covered [ GAP ] no test\n```\n' + flat(''))).toBe(false); + for (const status of ['cancelled', 'canceled', 'rejected', 'retracted', 'withdrawn', "'withdrawn'", '‘superseded’', '`no longer current`', '"not current"']) { + expect(diagram(flat('Legend [ OK ] covered [ GAP ] no test\nThis legend is ' + status + '.'))).toBe(false); + } + expect(diagram(flat('Legend [ OK ] covered [ GAP ] no test\n> An old note said: "This legend is withdrawn."'))).toBe(true); +}); + +test('text marker corrections grant only the final unambiguous owned row state', () => { + expect(diagram(flat().replace('success [ OK ]', 'success [ GAP ] -> [ OK ]'))).toBe(true); + expect(diagram(flat().replace('refunded [ GAP ]', 'refunded [ OK ] → [ GAP ]'))).toBe(true); + for (const [old, replacement] of [ + ['success [ OK ]', 'success COVERED [ OK ] → [ GAP ]'], + ['refunded [ GAP ]', 'refunded UNTESTED [ GAP ] -> [ OK ]'], + ['success [ OK ]', 'success COVERED [ OK ] [ GAP ]'], + ['refunded [ GAP ]', 'refunded [GAP] [ GAP ] [ OK ]'], + ['success [ OK ]', 'success not [ OK ]'], ['refunded [ GAP ]', 'refunded [ GAP ] is incorrect'], + ['success [ OK ]', 'success not covered [ OK ]'], ['refunded [ GAP ]', 'refunded no coverage gaps [ GAP ]'], + ]) expect(diagram(flat().replace(old!, replacement!))).toBe(false); +}); + +test('text coverage markers retain function, subtree, column and source ownership', () => { + for (const output of [ + flat().replace('processPayment', 'otherPayment'), flat().replace('refundPayment', 'otherRefund'), + flat().replace('├── happy', 'otherFunction()\n├── happy'), flat().replace('└── happy', 'otherFunction()\n└── happy'), + flat().replace('success [ OK ]', 'success ├── [ OK ]'), flat().replace('refunded [ GAP ]', 'refunded └── [ GAP ]'), + flat().split('\n').map(line => '> ' + line).join('\n'), '````markdown\n' + flat() + '\n````', 'Example:\n' + flat(), + ]) expect(diagram(output)).toBe(false); +}); +}); + +describe('coverage-checkbox-tail-av', () => { +const fixture = fixture_coverage_checkbox_tail_av; +const both={sourceRead:true,testsRead:true}, neither={sourceRead:false,testsRead:false}; +const fresh=(i=0)=>{const row=structuredClone(fixture.attempts[i]!) as any;return{row,use:row.result.transcript[1].message.content[0],ack:row.result.transcript[2].message.content[0]};}; +const reads=(mutate:(s:ReturnType)=>void=()=>{})=>{const s=fresh();mutate(s);return coverageAuditReadEvidence(s.row.result.transcript,s.row.files);}; +const base='```text\nprocessPayment(amount, currency)\n├── happy return success [x]\nrefundPayment(paymentId, reason)\n└── happy return refunded [ ]\nLegend: [x] tested [ ] no test\n```'; +const diagram=(output:string)=>{const {row}=fresh();return coverageAuditVerdict({...row.result,output},row.files).diagram;}; + +test('both exact public attempts now provide their delivered files and owned checkbox diagram',()=>{ + expect(fixture.provenance.paidOutcomesReclassified).toBe(false); + for(const row of fixture.attempts){expect(row.provenance.recordedPassed).toBe(false);expect(coverageAuditVerdict(row.result as any,row.files)).toEqual({...both,diagram:true,passed:true,failures:[]});} +}); + +test('mixed display tail accepts only the two ordered owned reads and literal separators',()=>{ + expect(reads()).toEqual(both); + for(const revision of ['HEAD','HEAD~1','main..HEAD'])expect(reads(s=>{s.use.input.command=s.use.input.command.replace('main..HEAD',revision).replace('diff main','diff '+revision);})).toEqual(both); + expect(reads(s=>{s.use.input.command=s.use.input.command.replaceAll('echo ====','echo ----');s.ack.content=s.ack.content.replaceAll('====','----');})).toEqual(both); + expect(reads(s=>{s.use.input.command=s.use.input.command.replace('src/billing.ts',"'src/billing.ts'").replace('test/billing.test.ts','"test/billing.test.ts"');})).toEqual(both); + expect(reads(s=>{s.ack.content=[{type:'text',text:s.ack.content}];})).toEqual(both); + expect(reads(s=>{s.ack.content+='\nabc1234 harmless commit subject\n src/billing.ts | 2 ++\n 1 file changed, 2 insertions(+)';})).toEqual(both); +}); + +test.each([ + 'false && cat -n src/billing.ts && cat -n test/billing.test.ts && git log --oneline main..HEAD; git diff main --stat', + 'cat -n src/billing.ts && false && cat -n test/billing.test.ts && git log --oneline main..HEAD; git diff main --stat', + 'cat -n fake.ts && cat -n test/billing.test.ts && git log --oneline main..HEAD; git diff main --stat', + 'cat -n src/billing.ts && cat -n src/billing.ts && git log --oneline main..HEAD; git diff main --stat', + 'cat -n src/billing.ts && cat -n test/billing.test.ts; git log --oneline main..HEAD; git diff main --stat', + 'cat -n src/billing.ts; cat -n test/billing.test.ts && git log --oneline main..HEAD; git diff main --stat', +])('a failed, skipped, duplicate or unrelated read cannot borrow display-tail success: %s',command=>{ + expect(reads(s=>{s.use.input.command=command;})).toEqual(neither); +}); + +test.each([ + ['git log --oneline main..HEAD','git log --format=%B main..HEAD'], + ['git log --oneline main..HEAD','git log --oneline --output=src/billing.ts main..HEAD'], + ['git log --oneline main..HEAD','git -c core.pager=evil log --oneline main..HEAD'], + ['git diff main --stat','git diff --ext-diff main --stat'], + ['git diff main --stat','git diff --no-index main --stat'], + ['git diff main --stat','git diff main --stat; echo extra'], + ['git diff main --stat','git diff main --stat > src/billing.ts'], + ['echo ====','echo replacement'],['echo ====','printf ===='],['echo ====','echo -e "\\nreplacement"'], + ['cat -n src/billing.ts','cat -n src/billing.ts > test/billing.test.ts'], + ['cat -n src/billing.ts','cat -n $(echo src/billing.ts)'], + ['cat -n src/billing.ts','rm src/billing.ts'], +])('replacement output or mutation stays outside the closed display-tail form', (old,next)=>{ + expect(reads(s=>{s.use.input.command=s.use.input.command.replace(old,next);})).toEqual(neither); +}); + +test('the native result must deliver exact ordered complete reads, even when Git hides a prefix failure',()=>{ + for(const mutate of [ + (s:ReturnType)=>{s.ack.is_error=true;}, + (s:ReturnType)=>{s.ack.content='cat: src/billing.ts: No such file\n'+s.ack.content;}, + (s:ReturnType)=>{s.ack.content=s.ack.content.replace('====\n','====\ncat: test/billing.test.ts: Permission denied\n');}, + (s:ReturnType)=>{s.ack.content=s.row.files.source.content;}, + (s:ReturnType)=>{s.ack.content=s.row.files.tests.content;}, + (s:ReturnType)=>{s.ack.content=s.ack.content.replace("return { status: 'success', amount, currency };","return undefined;");}, + (s:ReturnType)=>{const parts=s.ack.content.split('====');s.ack.content=parts[1]+'===='+parts[0]+'====';}, + (s:ReturnType)=>{s.row.result.transcript[2].session_id='foreign';}, + (s:ReturnType)=>{s.row.result.transcript[2].parent_tool_use_id='child';}, + (s:ReturnType)=>{s.ack.tool_use_id='foreign';}, + (s:ReturnType)=>{s.row.result.transcript.push(structuredClone(s.row.result.transcript[2]));}, + ])expect(reads(mutate)).toEqual(neither); +}); + +test('checkbox legends permit current synonyms, pair order, above/below placement and case',()=>{ + for(const legend of ['Legend: [x] tested [ ] no test','Legend [x] covered by an existing test; [ ] no test reaches this path','Legend: [ ] untested | [X] covered'])expect(diagram(base.replace('Legend: [x] tested [ ] no test',legend))).toBe(true); + expect(diagram(base.replaceAll('[x]','[X]'))).toBe(true); + expect(diagram(base.replace('Legend: [x] tested [ ] no test\n','').replace('processPayment','Legend: [x] tested [ ] no test\nprocessPayment'))).toBe(true); +}); + +test.each(['','> Legend: [x] tested [ ] no test','"Legend: [x] tested [ ] no test"','Source: Legend: [x] tested [ ] no test','If approved, Legend: [x] tested [ ] no test','Legend: [x] untested [ ] covered','Legend: [x] tested [ ] covered','Legend: [x] tested [x] no test','Legend: [x] tested [ ] no test except refunds','Legend: [x] tested [ ] no test\nLegend: [x] untested [ ] covered'])('missing or contradictory checkbox key gives no diagram coverage: %s',legend=>{ + expect(diagram(base.replace('Legend: [x] tested [ ] no test',legend))).toBe(false); +}); + +test('checkbox meanings cannot come from another block, stale key, or source declaration',()=>{ + expect(diagram('```\nLegend: [x] tested [ ] no test\n```\n'+base.replace('Legend: [x] tested [ ] no test\n',''))).toBe(false); + for(const status of ['withdrawn','`no longer current`',"'superseded'",'“rejected”'])for(const boundary of ['\n','\nAssessment complete; '])expect(diagram(base.replace('\n```',boundary+'This legend is '+status+'.\n```'))).toBe(false); + for(const statement of [' This legend is withdrawn.','**This legend** is `no longer current`.','This legend applies only if approved.'])expect(diagram(base.replace('\n```','\n'+statement+'\n```'))).toBe(false); + expect(diagram(base.replace('\n```','\nEarlier reviewer said "This legend is withdrawn."\n```'))).toBe(true); + expect(diagram(base.replace('\n```','\n> Earlier note; This legend is withdrawn.\n```'))).toBe(true); + for(const prefix of ['Source:','Historical note:','Hypothetical:'])expect(diagram(base.replace('Legend:',prefix+'\nLegend:'))).toBe(false); +}); + +test('checkbox states retain final correction, function subtree and column ownership',()=>{ + expect(diagram(base.replace('success [x]','success [ ] -> [x]'))).toBe(true); + expect(diagram(base.replace('refunded [ ]','refunded [x] → [ ]'))).toBe(true); + for(const [old,next]of [['success [x]','success [x] [ ]'],['refunded [ ]','refunded [ ] [x]'],['success [x]','success not [x]'],['refunded [ ]','refunded [ ] is incorrect'],['success [x]','success never covered [x]'],['refunded [ ]','refunded no coverage gaps [ ]'],['success [x]','success [x] -> [ ]'],['refunded [ ]','refunded [ ] → [x]'],['success [x]','success ├── [x]'],['refunded [ ]','refunded └── [ ]']])expect(diagram(base.replace(old!,next!))).toBe(false); + for(const name of ['processPayment','refundPayment'])expect(diagram(base.replace(name,'unrelated'))).toBe(false); + expect(diagram(base.replace('└── happy','otherFunction()\n└── happy'))).toBe(false); + expect(diagram(base.split('\n').map(l=>'> '+l).join('\n'))).toBe(false); + expect(diagram('````markdown\n'+base+'\n````')).toBe(false); + expect(diagram('Example:\n'+base)).toBe(false); +}); +}); + +describe('coverage-diagram-legend-as', () => { +const captured = captured_coverage_diagram_legend_as; +const billing = fixture; + +function verdict(output: string, index = 0) { + const row = captured.rows[index]!; + return coverageAuditVerdict({ ...row.result, output } as any, { + cwd: row.cwd, + source: { path: row.cwd + '/src/billing.ts', content: billing.files.source }, + tests: { path: row.cwd + '/test/billing.test.ts', content: billing.files.tests }, + }); +} +const diagram = (output: string) => verdict(output).diagram; +const flat = (legend = 'Legend: [✔] tested [✘] GAP (no test)') => '```text\n' + legend + '\nprocessPayment(amount, currency)\n├──► return success [✔]\nrefundPayment(paymentId, reason)\n└──► return refunded [✘]\n```'; + +test('both exact public outputs contain the seeded diagram and retain actual native file delivery', () => { + for (let i = 0; i < captured.rows.length; i++) { + expect(verdict(captured.rows[i]!.result.output, i)).toEqual({ sourceRead: true, testsRead: true, diagram: true, passed: true, failures: [] }); + } + expect(captured.provenance.originalAttemptOutcomes).toEqual(['failed', 'failed']); + expect(captured.provenance.paidOutcomesReclassified).toBe(false); +}); + +test('closed legend annotations preserve the same two meanings and arrow branch ownership', () => { + for (const legend of ['Legend: [✔] tested [✘] GAP', 'Legend: [✔] tested [✘] GAP (no test)', 'Legend: [✔] tested [✘] GAP ──► branch', 'Legend: [✔] tested [✘] GAP (no test) ──► branch']) { + expect(diagram(flat(legend))).toBe(true); + expect(diagram(flat(legend).replace(/^([├└]─+)►/gm, '$1'))).toBe(true); + expect(diagram(flat(legend).replaceAll('✔', '✓').replaceAll('✘', '✗'))).toBe(true); + } +}); + +test('extra legend explanations cannot invert, qualify or fabricate coverage meanings', () => { + for (const legend of ['', 'Legend: [✔] GAP [✘] tested', 'Legend: [✔] tested [✘] tested', 'Legend: [✔] tested [✘] GAP (not a gap)', 'Legend: [✔] tested [✘] GAP except refunds', 'Legend: [✔] tested [✘] GAP [✘] covered', 'Example: [✔] tested [✘] GAP', 'Legend: not [✔] tested [✘] GAP', 'Legend: [✔] tested [✘] GAP ──► covered']) { + expect(diagram(flat(legend))).toBe(false); + } +}); + +test('a branch status correction supplies its final state and ambiguous markers supply neither', () => { + expect(diagram(flat().replace('return refunded [✘]', 'return refunded [✔]→[✘]'))).toBe(true); + expect(diagram(flat().replace('return success [✔]', 'return success [✘]->[✔]'))).toBe(true); + expect(diagram(flat().replace('return success [✔]', 'return success [✔]→[✘]'))).toBe(false); + expect(diagram(flat().replace('return refunded [✘]', 'return refunded [✘]→[✔]'))).toBe(false); + expect(diagram(flat().replace('return success [✔]', 'return success [✔] [✘]'))).toBe(false); + expect(diagram(flat().replace('return refunded [✘]', 'return refunded [✘] [✔]'))).toBe(false); +}); + +test('literal labels cannot override a final or ambiguous bracketed symbol state', () => { + for (const [old, replacement] of [ + ['return success [✔]', 'return success TESTED [✔]→[✘]'], + ['return refunded [✘]', 'return refunded UNTESTED [✘]→[✔]'], + ['return success [✔]', 'return success TESTED [✔] [✘]'], + ['return refunded [✘]', 'return refunded [GAP] [✘] [✔]'], + ]) expect(diagram(flat().replace(old!, replacement!))).toBe(false); +}); + +test('each seeded function must own its own branch and legend in the same current diagram', () => { + for (const output of [ + flat().replace('refundPayment', 'otherRefund'), flat().replace('processPayment', 'otherPayment'), + flat().replace('├──► return success [✔]', 'unrelatedHelper()\n├──► return success [✔]'), + flat().replace('└──► return refunded [✘]', 'unrelatedHelper()\n└──► return refunded [✘]'), + flat().split('\n').map(line => '> ' + line).join('\n'), '````markdown\n' + flat() + '\n````', + 'Example:\n' + flat(), flat().replace('[✔] tested [✘] GAP (no test)', '[✔] tested (not covered) [✘] GAP'), + '```text\nLegend: [✔] tested [✘] GAP\n```\n' + flat(''), + ]) expect(diagram(output)).toBe(false); +}); + +test('successful diagram parsing cannot replace successful capture or native file delivery', () => { + const row = captured.rows[0]!; + const files = { cwd: row.cwd, source: { path: row.cwd + '/src/billing.ts', content: billing.files.source }, tests: { path: row.cwd + '/test/billing.test.ts', content: billing.files.tests } }; + for (const mutate of [ + (r: any) => { r.exitReason = 'timeout'; }, (r: any) => { r.browseErrors = ['read failed']; }, + (r: any) => { r.transcript = []; }, (r: any) => { r.transcript[2].message.content[0].is_error = true; }, + (r: any) => { r.transcript[2].message.content[0].content = 'Both filenames were read'; }, + ]) { + const result = structuredClone(row.result); mutate(result); + const checked = coverageAuditVerdict(result as any, files); + expect(checked.diagram).toBe(true); expect(checked.passed).toBe(false); + } +}); +}); + +describe('coverage-shell-display-aq', () => { +const path = posix; +const fixture = fixture_coverage_shell_display_aq; +const billing = billing_coverage_shell_display_aq; +function replay(row: typeof fixture.rows[number], command?: string) { + const transcript = structuredClone(row.transcript) as any[]; + if (command !== undefined) transcript[1].message.content[0].input.command = command; + const cwd = transcript[0].cwd; + return coverageAuditReadEvidence(transcript, { + cwd, source: { path: path.join(cwd, 'src/billing.ts'), content: billing.files.source }, + tests: { path: path.join(cwd, 'test/billing.test.ts'), content: billing.files.tests }, + }); +} +const command = (row: typeof fixture.rows[number]) => (row.transcript[1] as any).message.content[0].input.command as string; + +describe('coverage reads with neighboring display commands', () => { + test('both exact failed AQ attempts delivered source and tests in their acknowledged Bash result', () => { + expect(fixture.provenance.actualPassedCases).toBe(0); + for (const row of fixture.rows) expect(replay(row)).toEqual({ sourceRead: true, testsRead: true }); + }); + + test('a literal grep range and numeric Git log count do not own the delivered file bytes', () => { + const first = fixture.rows[0]!, second = fixture.rows[1]!; + expect(replay(first, command(first).replace('head -40', 'head -25'))).toEqual({ sourceRead: true, testsRead: true }); + expect(replay(second, command(second).replace('log --oneline -3', 'log --oneline -12'))).toEqual({ sourceRead: true, testsRead: true }); + }); + + test.each([ + ['awk action', (s: string) => s.replace("awk '/^### 3\\. Test review/,/^### 4\\./'", "awk 'BEGIN { system(\"cat fake\") }'")], + ['awk output redirection', (s: string) => s.replace("awk '/^### 3\\. Test review/,/^### 4\\./'", "awk '/x/ { print > \"src/billing.ts\" }'")], + ['shell substitution', (s: string) => s.replace('grep -n', 'grep -n "$(cat fake)"')], + ['backtick execution', (s: string) => s.replace('grep -n', 'grep -n `cat fake`')], + ['quoted injected command', (s: string) => s.replace('grep -n', 'grep -n "x"; printf fake; grep -n')], + ['read hidden in a conditional', (s: string) => s.replace('cat -n src/billing.ts', 'false && cat -n src/billing.ts')], + ['source-only filename', (s: string) => s.replace('cat -n src/billing.ts', "echo 'cat -n src/billing.ts'")], + ] as const)('%s cannot borrow source read evidence', (_, mutate) => { + const row = fixture.rows[0]!; + expect(mutate(command(row))).not.toBe(command(row)); + expect(replay(row, mutate(command(row))).sourceRead).toBe(false); + }); + + test.each([ + 'git log --output=src/billing.ts -3', + 'git log --ext-diff -3', + 'git log --format=%x00 -3', + 'git log -3; printf fake', + ])('unsupported Git command %s cannot borrow delivery', git => { + const row = fixture.rows[1]!; + expect(replay(row, command(row).replace('git log --oneline -3', git)).sourceRead).toBe(false); + }); + + test.each(['-f/tmp/other.awk', "'-f/tmp/other.awk'", "'--source=BEGIN {print \"fake\"}'"])( + 'awk input %s cannot introduce another program', operand => { + const row = fixture.rows[0]!; + const changed = command(row).replace("Test review/,/^### 4\\./' plan-eng-review/sections/review-sections.md", "Test review/,/^### 4\\./' " + operand); + expect(changed).not.toBe(command(row)); + expect(replay(row, changed).sourceRead).toBe(false); + }); + + test('successful command identity still requires the complete file and paired parent result', () => { + for (const row of fixture.rows) { + const missing = structuredClone(row) as any; + missing.transcript[2].message.content[0].content = 'src/billing.ts and test/billing.test.ts were read'; + expect(replay(missing)).toEqual({ sourceRead: false, testsRead: false }); + const failed = structuredClone(row) as any; + failed.transcript[2].message.content[0].is_error = true; + expect(replay(failed)).toEqual({ sourceRead: false, testsRead: false }); + } + }); +}); +}); diff --git a/test/coverage-audit-shell-legend-at.test.ts b/test/coverage-audit-shell-legend-at.test.ts deleted file mode 100644 index 439b1c8a7..000000000 --- a/test/coverage-audit-shell-legend-at.test.ts +++ /dev/null @@ -1,121 +0,0 @@ -import { expect, test } from 'bun:test'; -import captured from './fixtures/coverage-audit-shell-legend-at.json'; -import { coverageAuditReadEvidence, coverageAuditVerdict } from './helpers/coverage-audit-evidence'; -import { E2E_TOUCHFILES, selectTests } from './helpers/touchfiles'; - -const both = { sourceRead: true, testsRead: true }; -const neither = { sourceRead: false, testsRead: false }; -function owned(index: number) { - const row = structuredClone(captured[index]!) as any; - const useEvent = row.result.transcript.find((event: any) => event.message?.content.some((block: any) => - block.type === 'tool_use' && block.name === 'Bash' && block.input.command.includes('cat -n src/billing.ts'))); - const use = useEvent.message.content.find((block: any) => block.type === 'tool_use' && block.name === 'Bash' && block.input.command.includes('cat -n src/billing.ts')); - const resultEvent = row.result.transcript.find((event: any) => event.message?.content.some((block: any) => block.type === 'tool_result' && block.tool_use_id === use.id)); - row.result.transcript = [row.result.transcript.find((event: any) => event.type === 'system' && event.subtype === 'init'), useEvent, resultEvent]; - return { row, use, resultEvent, delivered: resultEvent.message.content.find((block: any) => block.tool_use_id === use.id) }; -} -function reads(index: number, mutate?: (s: ReturnType) => void) { - const s = owned(index); mutate?.(s); - return coverageAuditReadEvidence(s.row.result.transcript, s.row.files); -} -const flat = (legend = 'Legend [ OK ] covered [ GAP ] no test') => - '```text\nprocessPayment(amount, currency)\n├── happy path return success [ OK ]\nrefundPayment(paymentId, reason)\n└── happy path return refunded [ GAP ]\n' + legend + '\n```'; -const diagram = (output: string) => coverageAuditVerdict({ ...captured[1]!.result, output } as any, captured[1]!.files).diagram; - -test('exact public AT first, retry and engineering outputs retain all required native evidence', () => { - expect(captured.map(row => row.recordedPassed)).toEqual([false, false, true]); - for (const row of captured) expect(coverageAuditVerdict(row.result as any, row.files)).toEqual({ ...both, diagram: true, passed: true, failures: [] }); - expect(reads(0)).toEqual(both); expect(reads(1)).toEqual(both); -}); - -test('literal grep display options and filename captions do not own source bytes', () => { - for (const flags of ['-n', '-n -i', '-n -B1 -A200', '-n -i -B3 -A40']) { - expect(reads(0, s => { s.use.input.command = s.use.input.command.replace('-n -i -B3 -A40', flags); })).toEqual(both); - } - for (const replace of ['echo \'=== another-file.md ===\'', 'echo "--- src/billing.ts ---"', 'echo']) { - expect(reads(1, s => { s.use.input.command = s.use.input.command.replace('echo "=== testing.md ==="', replace); })).toEqual(both); - } - expect(reads(1, s => { s.use.input.command = s.use.input.command.replace('git diff main --stat', 'git diff HEAD~1 --stat'); })).toEqual(both); -}); - -test('escaped grep patterns keep a closed flag and literal operand grammar', () => { - for (const replacement of ['-n -i -B3 -A40 -f other', '-n -i --include=*', '-n -B-1', '-n -A100000', '-n -i -B3 -A40; false']) { - expect(reads(0, s => { s.use.input.command = s.use.input.command.replace('-n -i -B3 -A40', replacement); })).toEqual(neither); - } - for (const operand of ['-f/tmp/foreign', '"-f/tmp/foreign"', 'review/SKILL.md --include=*']) { - expect(reads(0, s => { s.use.input.command = s.use.input.command.replace('review/SKILL.md |', operand + ' |'); })).toEqual(neither); - } -}); - -test('successful conditional display paths reject execution, substitutions and hidden failure', () => { - for (const replacement of [ - 'echo -e "=== testing.md ==="', 'printf "=== testing.md ==="', 'echo "$(cat fake)"', 'echo `cat fake`', - 'echo "=== testing.md ==="; false', 'false || echo "=== testing.md ==="', 'unknown', - 'echo "cat -n src/billing.ts"', 'echo "=== testing.md ===\\nreplacement"', - 'cd ../sibling', 'env PATH=/tmp cat fake', 'echo "=== testing.md ===" > src/billing.ts', - ]) expect(reads(1, s => { s.use.input.command = s.use.input.command.replace('echo "=== testing.md ==="', replacement); })).toEqual(neither); - for (const command of ['git diff --ext-diff --stat', 'git diff main --output=src/billing.ts --stat', 'git -c core.pager=evil diff main --stat', 'git diff --no-index main --stat', 'git diff main --stat || echo ok']) { - expect(reads(1, s => { s.use.input.command = s.use.input.command.replace('git diff main --stat', command); })).toEqual(neither); - } -}); - -test('a valid display path still requires one complete successful owned delivery', () => { - for (const index of [0, 1]) for (const mutate of [ - (s: ReturnType) => { s.delivered.is_error = true; }, - (s: ReturnType) => { s.delivered.content = 'src/billing.ts and test/billing.test.ts were read'; }, - (s: ReturnType) => { s.delivered.content = s.row.files.source.content.slice(0, 80); }, - (s: ReturnType) => { s.resultEvent.session_id = 'foreign'; }, - (s: ReturnType) => { s.resultEvent.parent_tool_use_id = 'child'; }, - (s: ReturnType) => { s.delivered.tool_use_id = 'foreign'; }, - (s: ReturnType) => { s.row.result.transcript.push(structuredClone(s.resultEvent)); }, - (s: ReturnType) => { s.use.input.command = s.use.input.command.replace('cat -n src/billing.ts', 'echo src/billing.ts').replace('cat -n test/billing.test.ts', 'echo test/billing.test.ts'); }, - ]) expect(reads(index, mutate)).toEqual(neither); - expect(reads(1, s => { s.delivered.content = s.row.files.source.content; })).toEqual({ sourceRead: true, testsRead: false }); - expect(reads(1, s => { s.delivered.content = s.row.files.tests.content; })).toEqual({ sourceRead: false, testsRead: true }); -}); - -test('text statuses use the declared local meanings with whitespace and either pair order', () => { - for (const legend of ['Legend [ OK ] covered [ GAP ] no test', 'Legend: [OK] tested | [GAP] untested', 'Legend: [ GAP ] no test; [ OK ] covered']) expect(diagram(flat(legend))).toBe(true); - expect(diagram(flat().replaceAll('[ OK ]', '[OK]').replaceAll('[ GAP ]', '[GAP]'))).toBe(true); - expect(diagram(flat().replace('Legend [ OK ] covered [ GAP ] no test\n', '').replace('processPayment', 'Legend [ OK ] covered [ GAP ] no test\nprocessPayment'))).toBe(true); -}); - -test('missing, malformed, contradictory or foreign text legends cannot grant coverage', () => { - for (const legend of ['', '> Legend [ OK ] covered [ GAP ] no test', '"Legend [ OK ] covered [ GAP ] no test"', - 'Example: Legend [ OK ] covered [ GAP ] no test', 'If enabled, Legend [ OK ] covered [ GAP ] no test', - 'Legend [ OK ] no test [ GAP ] covered', 'Legend [ OK ] covered [ GAP ] covered', - 'Legend [ OK ] covered [ OK ] no test', 'Legend [ OK ] covered [ GAP ] no test except refunds', - 'Legend [ OK ] covered [ GAP ] no test\nLegend [ OK ] no test [ GAP ] covered', - ]) expect(diagram(flat(legend))).toBe(false); - expect(diagram('```text\nLegend [ OK ] covered [ GAP ] no test\n```\n' + flat(''))).toBe(false); - for (const status of ['cancelled', 'canceled', 'rejected', 'retracted', 'withdrawn', "'withdrawn'", '‘superseded’', '`no longer current`', '"not current"']) { - expect(diagram(flat('Legend [ OK ] covered [ GAP ] no test\nThis legend is ' + status + '.'))).toBe(false); - } - expect(diagram(flat('Legend [ OK ] covered [ GAP ] no test\n> An old note said: "This legend is withdrawn."'))).toBe(true); -}); - -test('text marker corrections grant only the final unambiguous owned row state', () => { - expect(diagram(flat().replace('success [ OK ]', 'success [ GAP ] -> [ OK ]'))).toBe(true); - expect(diagram(flat().replace('refunded [ GAP ]', 'refunded [ OK ] → [ GAP ]'))).toBe(true); - for (const [old, replacement] of [ - ['success [ OK ]', 'success COVERED [ OK ] → [ GAP ]'], - ['refunded [ GAP ]', 'refunded UNTESTED [ GAP ] -> [ OK ]'], - ['success [ OK ]', 'success COVERED [ OK ] [ GAP ]'], - ['refunded [ GAP ]', 'refunded [GAP] [ GAP ] [ OK ]'], - ['success [ OK ]', 'success not [ OK ]'], ['refunded [ GAP ]', 'refunded [ GAP ] is incorrect'], - ['success [ OK ]', 'success not covered [ OK ]'], ['refunded [ GAP ]', 'refunded no coverage gaps [ GAP ]'], - ]) expect(diagram(flat().replace(old!, replacement!))).toBe(false); -}); - -test('text coverage markers retain function, subtree, column and source ownership', () => { - for (const output of [ - flat().replace('processPayment', 'otherPayment'), flat().replace('refundPayment', 'otherRefund'), - flat().replace('├── happy', 'otherFunction()\n├── happy'), flat().replace('└── happy', 'otherFunction()\n└── happy'), - flat().replace('success [ OK ]', 'success ├── [ OK ]'), flat().replace('refunded [ GAP ]', 'refunded └── [ GAP ]'), - flat().split('\n').map(line => '> ' + line).join('\n'), '````markdown\n' + flat() + '\n````', 'Example:\n' + flat(), - ]) expect(diagram(output)).toBe(false); -}); - -test('the regression fixture and controls select both paid coverage owners', () => { - for (const file of ['test/coverage-audit-shell-legend-at.test.ts', 'test/fixtures/coverage-audit-shell-legend-at.json']) expect(selectTests([file], E2E_TOUCHFILES, []).selected.sort()).toEqual(['plan-eng-coverage-audit', 'review-coverage-audit']); -}); diff --git a/test/coverage-checkbox-tail-av.test.ts b/test/coverage-checkbox-tail-av.test.ts deleted file mode 100644 index 16adc87d1..000000000 --- a/test/coverage-checkbox-tail-av.test.ts +++ /dev/null @@ -1,103 +0,0 @@ -import {expect,test} from 'bun:test'; -import fixture from './fixtures/coverage-checkbox-tail-av.json'; -import {coverageAuditReadEvidence,coverageAuditVerdict} from './helpers/coverage-audit-evidence'; -import {E2E_TOUCHFILES,LLM_JUDGE_TOUCHFILES,selectTests} from './helpers/touchfiles'; -const both={sourceRead:true,testsRead:true}, neither={sourceRead:false,testsRead:false}; -const fresh=(i=0)=>{const row=structuredClone(fixture.attempts[i]!) as any;return{row,use:row.result.transcript[1].message.content[0],ack:row.result.transcript[2].message.content[0]};}; -const reads=(mutate:(s:ReturnType)=>void=()=>{})=>{const s=fresh();mutate(s);return coverageAuditReadEvidence(s.row.result.transcript,s.row.files);}; -const base='```text\nprocessPayment(amount, currency)\n├── happy return success [x]\nrefundPayment(paymentId, reason)\n└── happy return refunded [ ]\nLegend: [x] tested [ ] no test\n```'; -const diagram=(output:string)=>{const {row}=fresh();return coverageAuditVerdict({...row.result,output},row.files).diagram;}; - -test('both exact public attempts now provide their delivered files and owned checkbox diagram',()=>{ - expect(fixture.provenance.paidOutcomesReclassified).toBe(false); - for(const row of fixture.attempts){expect(row.provenance.recordedPassed).toBe(false);expect(coverageAuditVerdict(row.result as any,row.files)).toEqual({...both,diagram:true,passed:true,failures:[]});} -}); - -test('mixed display tail accepts only the two ordered owned reads and literal separators',()=>{ - expect(reads()).toEqual(both); - for(const revision of ['HEAD','HEAD~1','main..HEAD'])expect(reads(s=>{s.use.input.command=s.use.input.command.replace('main..HEAD',revision).replace('diff main','diff '+revision);})).toEqual(both); - expect(reads(s=>{s.use.input.command=s.use.input.command.replaceAll('echo ====','echo ----');s.ack.content=s.ack.content.replaceAll('====','----');})).toEqual(both); - expect(reads(s=>{s.use.input.command=s.use.input.command.replace('src/billing.ts',"'src/billing.ts'").replace('test/billing.test.ts','"test/billing.test.ts"');})).toEqual(both); - expect(reads(s=>{s.ack.content=[{type:'text',text:s.ack.content}];})).toEqual(both); - expect(reads(s=>{s.ack.content+='\nabc1234 harmless commit subject\n src/billing.ts | 2 ++\n 1 file changed, 2 insertions(+)';})).toEqual(both); -}); - -test.each([ - 'false && cat -n src/billing.ts && cat -n test/billing.test.ts && git log --oneline main..HEAD; git diff main --stat', - 'cat -n src/billing.ts && false && cat -n test/billing.test.ts && git log --oneline main..HEAD; git diff main --stat', - 'cat -n fake.ts && cat -n test/billing.test.ts && git log --oneline main..HEAD; git diff main --stat', - 'cat -n src/billing.ts && cat -n src/billing.ts && git log --oneline main..HEAD; git diff main --stat', - 'cat -n src/billing.ts && cat -n test/billing.test.ts; git log --oneline main..HEAD; git diff main --stat', - 'cat -n src/billing.ts; cat -n test/billing.test.ts && git log --oneline main..HEAD; git diff main --stat', -])('a failed, skipped, duplicate or unrelated read cannot borrow display-tail success: %s',command=>{ - expect(reads(s=>{s.use.input.command=command;})).toEqual(neither); -}); - -test.each([ - ['git log --oneline main..HEAD','git log --format=%B main..HEAD'], - ['git log --oneline main..HEAD','git log --oneline --output=src/billing.ts main..HEAD'], - ['git log --oneline main..HEAD','git -c core.pager=evil log --oneline main..HEAD'], - ['git diff main --stat','git diff --ext-diff main --stat'], - ['git diff main --stat','git diff --no-index main --stat'], - ['git diff main --stat','git diff main --stat; echo extra'], - ['git diff main --stat','git diff main --stat > src/billing.ts'], - ['echo ====','echo replacement'],['echo ====','printf ===='],['echo ====','echo -e "\\nreplacement"'], - ['cat -n src/billing.ts','cat -n src/billing.ts > test/billing.test.ts'], - ['cat -n src/billing.ts','cat -n $(echo src/billing.ts)'], - ['cat -n src/billing.ts','rm src/billing.ts'], -])('replacement output or mutation stays outside the closed display-tail form', (old,next)=>{ - expect(reads(s=>{s.use.input.command=s.use.input.command.replace(old,next);})).toEqual(neither); -}); - -test('the native result must deliver exact ordered complete reads, even when Git hides a prefix failure',()=>{ - for(const mutate of [ - (s:ReturnType)=>{s.ack.is_error=true;}, - (s:ReturnType)=>{s.ack.content='cat: src/billing.ts: No such file\n'+s.ack.content;}, - (s:ReturnType)=>{s.ack.content=s.ack.content.replace('====\n','====\ncat: test/billing.test.ts: Permission denied\n');}, - (s:ReturnType)=>{s.ack.content=s.row.files.source.content;}, - (s:ReturnType)=>{s.ack.content=s.row.files.tests.content;}, - (s:ReturnType)=>{s.ack.content=s.ack.content.replace("return { status: 'success', amount, currency };","return undefined;");}, - (s:ReturnType)=>{const parts=s.ack.content.split('====');s.ack.content=parts[1]+'===='+parts[0]+'====';}, - (s:ReturnType)=>{s.row.result.transcript[2].session_id='foreign';}, - (s:ReturnType)=>{s.row.result.transcript[2].parent_tool_use_id='child';}, - (s:ReturnType)=>{s.ack.tool_use_id='foreign';}, - (s:ReturnType)=>{s.row.result.transcript.push(structuredClone(s.row.result.transcript[2]));}, - ])expect(reads(mutate)).toEqual(neither); -}); - -test('checkbox legends permit current synonyms, pair order, above/below placement and case',()=>{ - for(const legend of ['Legend: [x] tested [ ] no test','Legend [x] covered by an existing test; [ ] no test reaches this path','Legend: [ ] untested | [X] covered'])expect(diagram(base.replace('Legend: [x] tested [ ] no test',legend))).toBe(true); - expect(diagram(base.replaceAll('[x]','[X]'))).toBe(true); - expect(diagram(base.replace('Legend: [x] tested [ ] no test\n','').replace('processPayment','Legend: [x] tested [ ] no test\nprocessPayment'))).toBe(true); -}); - -test.each(['','> Legend: [x] tested [ ] no test','"Legend: [x] tested [ ] no test"','Source: Legend: [x] tested [ ] no test','If approved, Legend: [x] tested [ ] no test','Legend: [x] untested [ ] covered','Legend: [x] tested [ ] covered','Legend: [x] tested [x] no test','Legend: [x] tested [ ] no test except refunds','Legend: [x] tested [ ] no test\nLegend: [x] untested [ ] covered'])('missing or contradictory checkbox key gives no diagram coverage: %s',legend=>{ - expect(diagram(base.replace('Legend: [x] tested [ ] no test',legend))).toBe(false); -}); - -test('checkbox meanings cannot come from another block, stale key, or source declaration',()=>{ - expect(diagram('```\nLegend: [x] tested [ ] no test\n```\n'+base.replace('Legend: [x] tested [ ] no test\n',''))).toBe(false); - for(const status of ['withdrawn','`no longer current`',"'superseded'",'“rejected”'])for(const boundary of ['\n','\nAssessment complete; '])expect(diagram(base.replace('\n```',boundary+'This legend is '+status+'.\n```'))).toBe(false); - for(const statement of [' This legend is withdrawn.','**This legend** is `no longer current`.','This legend applies only if approved.'])expect(diagram(base.replace('\n```','\n'+statement+'\n```'))).toBe(false); - expect(diagram(base.replace('\n```','\nEarlier reviewer said "This legend is withdrawn."\n```'))).toBe(true); - expect(diagram(base.replace('\n```','\n> Earlier note; This legend is withdrawn.\n```'))).toBe(true); - for(const prefix of ['Source:','Historical note:','Hypothetical:'])expect(diagram(base.replace('Legend:',prefix+'\nLegend:'))).toBe(false); -}); - -test('checkbox states retain final correction, function subtree and column ownership',()=>{ - expect(diagram(base.replace('success [x]','success [ ] -> [x]'))).toBe(true); - expect(diagram(base.replace('refunded [ ]','refunded [x] → [ ]'))).toBe(true); - for(const [old,next]of [['success [x]','success [x] [ ]'],['refunded [ ]','refunded [ ] [x]'],['success [x]','success not [x]'],['refunded [ ]','refunded [ ] is incorrect'],['success [x]','success never covered [x]'],['refunded [ ]','refunded no coverage gaps [ ]'],['success [x]','success [x] -> [ ]'],['refunded [ ]','refunded [ ] → [x]'],['success [x]','success ├── [x]'],['refunded [ ]','refunded └── [ ]']])expect(diagram(base.replace(old!,next!))).toBe(false); - for(const name of ['processPayment','refundPayment'])expect(diagram(base.replace(name,'unrelated'))).toBe(false); - expect(diagram(base.replace('└── happy','otherFunction()\n└── happy'))).toBe(false); - expect(diagram(base.split('\n').map(l=>'> '+l).join('\n'))).toBe(false); - expect(diagram('````markdown\n'+base+'\n````')).toBe(false); - expect(diagram('Example:\n'+base)).toBe(false); -}); - -test('all three existing coverage consumers are selected without judge expansion',()=>{ - for(const file of ['test/coverage-checkbox-tail-av.test.ts','test/fixtures/coverage-checkbox-tail-av.json']){ - expect(selectTests([file],E2E_TOUCHFILES,[]).selected.sort()).toEqual(['plan-eng-coverage-audit','review-coverage-audit','ship-coverage-audit']); - expect(selectTests([file],LLM_JUDGE_TOUCHFILES,[]).selected).toEqual([]); - } -}); diff --git a/test/coverage-diagram-legend-as.test.ts b/test/coverage-diagram-legend-as.test.ts deleted file mode 100644 index f9085bc85..000000000 --- a/test/coverage-diagram-legend-as.test.ts +++ /dev/null @@ -1,85 +0,0 @@ -import { expect, test } from 'bun:test'; -import captured from './fixtures/coverage-diagram-legend-as.json'; -import billing from './fixtures/coverage-audit-ae.json'; -import { coverageAuditVerdict } from './helpers/coverage-audit-evidence'; -import { E2E_TOUCHFILES, selectTests } from './helpers/touchfiles'; - -function verdict(output: string, index = 0) { - const row = captured.rows[index]!; - return coverageAuditVerdict({ ...row.result, output } as any, { - cwd: row.cwd, - source: { path: row.cwd + '/src/billing.ts', content: billing.files.source }, - tests: { path: row.cwd + '/test/billing.test.ts', content: billing.files.tests }, - }); -} -const diagram = (output: string) => verdict(output).diagram; -const flat = (legend = 'Legend: [✔] tested [✘] GAP (no test)') => '```text\n' + legend + '\nprocessPayment(amount, currency)\n├──► return success [✔]\nrefundPayment(paymentId, reason)\n└──► return refunded [✘]\n```'; - -test('both exact public outputs contain the seeded diagram and retain actual native file delivery', () => { - for (let i = 0; i < captured.rows.length; i++) { - expect(verdict(captured.rows[i]!.result.output, i)).toEqual({ sourceRead: true, testsRead: true, diagram: true, passed: true, failures: [] }); - } - expect(captured.provenance.originalAttemptOutcomes).toEqual(['failed', 'failed']); - expect(captured.provenance.paidOutcomesReclassified).toBe(false); -}); - -test('closed legend annotations preserve the same two meanings and arrow branch ownership', () => { - for (const legend of ['Legend: [✔] tested [✘] GAP', 'Legend: [✔] tested [✘] GAP (no test)', 'Legend: [✔] tested [✘] GAP ──► branch', 'Legend: [✔] tested [✘] GAP (no test) ──► branch']) { - expect(diagram(flat(legend))).toBe(true); - expect(diagram(flat(legend).replace(/^([├└]─+)►/gm, '$1'))).toBe(true); - expect(diagram(flat(legend).replaceAll('✔', '✓').replaceAll('✘', '✗'))).toBe(true); - } -}); - -test('extra legend explanations cannot invert, qualify or fabricate coverage meanings', () => { - for (const legend of ['', 'Legend: [✔] GAP [✘] tested', 'Legend: [✔] tested [✘] tested', 'Legend: [✔] tested [✘] GAP (not a gap)', 'Legend: [✔] tested [✘] GAP except refunds', 'Legend: [✔] tested [✘] GAP [✘] covered', 'Example: [✔] tested [✘] GAP', 'Legend: not [✔] tested [✘] GAP', 'Legend: [✔] tested [✘] GAP ──► covered']) { - expect(diagram(flat(legend))).toBe(false); - } -}); - -test('a branch status correction supplies its final state and ambiguous markers supply neither', () => { - expect(diagram(flat().replace('return refunded [✘]', 'return refunded [✔]→[✘]'))).toBe(true); - expect(diagram(flat().replace('return success [✔]', 'return success [✘]->[✔]'))).toBe(true); - expect(diagram(flat().replace('return success [✔]', 'return success [✔]→[✘]'))).toBe(false); - expect(diagram(flat().replace('return refunded [✘]', 'return refunded [✘]→[✔]'))).toBe(false); - expect(diagram(flat().replace('return success [✔]', 'return success [✔] [✘]'))).toBe(false); - expect(diagram(flat().replace('return refunded [✘]', 'return refunded [✘] [✔]'))).toBe(false); -}); - -test('literal labels cannot override a final or ambiguous bracketed symbol state', () => { - for (const [old, replacement] of [ - ['return success [✔]', 'return success TESTED [✔]→[✘]'], - ['return refunded [✘]', 'return refunded UNTESTED [✘]→[✔]'], - ['return success [✔]', 'return success TESTED [✔] [✘]'], - ['return refunded [✘]', 'return refunded [GAP] [✘] [✔]'], - ]) expect(diagram(flat().replace(old!, replacement!))).toBe(false); -}); - -test('each seeded function must own its own branch and legend in the same current diagram', () => { - for (const output of [ - flat().replace('refundPayment', 'otherRefund'), flat().replace('processPayment', 'otherPayment'), - flat().replace('├──► return success [✔]', 'unrelatedHelper()\n├──► return success [✔]'), - flat().replace('└──► return refunded [✘]', 'unrelatedHelper()\n└──► return refunded [✘]'), - flat().split('\n').map(line => '> ' + line).join('\n'), '````markdown\n' + flat() + '\n````', - 'Example:\n' + flat(), flat().replace('[✔] tested [✘] GAP (no test)', '[✔] tested (not covered) [✘] GAP'), - '```text\nLegend: [✔] tested [✘] GAP\n```\n' + flat(''), - ]) expect(diagram(output)).toBe(false); -}); - -test('successful diagram parsing cannot replace successful capture or native file delivery', () => { - const row = captured.rows[0]!; - const files = { cwd: row.cwd, source: { path: row.cwd + '/src/billing.ts', content: billing.files.source }, tests: { path: row.cwd + '/test/billing.test.ts', content: billing.files.tests } }; - for (const mutate of [ - (r: any) => { r.exitReason = 'timeout'; }, (r: any) => { r.browseErrors = ['read failed']; }, - (r: any) => { r.transcript = []; }, (r: any) => { r.transcript[2].message.content[0].is_error = true; }, - (r: any) => { r.transcript[2].message.content[0].content = 'Both filenames were read'; }, - ]) { - const result = structuredClone(row.result); mutate(result); - const checked = coverageAuditVerdict(result as any, files); - expect(checked.diagram).toBe(true); expect(checked.passed).toBe(false); - } -}); - -test('new parser regression artifacts select both coverage audit owners', () => { - for (const file of ['test/coverage-diagram-legend-as.test.ts', 'test/fixtures/coverage-diagram-legend-as.json']) expect(selectTests([file], E2E_TOUCHFILES, []).selected.sort()).toEqual(['plan-eng-coverage-audit', 'review-coverage-audit']); -}); diff --git a/test/coverage-shell-display-aq.test.ts b/test/coverage-shell-display-aq.test.ts deleted file mode 100644 index 51425e170..000000000 --- a/test/coverage-shell-display-aq.test.ts +++ /dev/null @@ -1,72 +0,0 @@ -import { describe, expect, test } from 'bun:test'; -import { posix as path } from 'node:path'; -import fixture from './fixtures/coverage-shell-display-aq.json'; -import billing from './fixtures/coverage-audit-ae.json'; -import { coverageAuditReadEvidence } from './helpers/coverage-audit-evidence'; - -function replay(row: typeof fixture.rows[number], command?: string) { - const transcript = structuredClone(row.transcript) as any[]; - if (command !== undefined) transcript[1].message.content[0].input.command = command; - const cwd = transcript[0].cwd; - return coverageAuditReadEvidence(transcript, { - cwd, source: { path: path.join(cwd, 'src/billing.ts'), content: billing.files.source }, - tests: { path: path.join(cwd, 'test/billing.test.ts'), content: billing.files.tests }, - }); -} -const command = (row: typeof fixture.rows[number]) => (row.transcript[1] as any).message.content[0].input.command as string; - -describe('coverage reads with neighboring display commands', () => { - test('both exact failed AQ attempts delivered source and tests in their acknowledged Bash result', () => { - expect(fixture.provenance.actualPassedCases).toBe(0); - for (const row of fixture.rows) expect(replay(row)).toEqual({ sourceRead: true, testsRead: true }); - }); - - test('a literal grep range and numeric Git log count do not own the delivered file bytes', () => { - const first = fixture.rows[0]!, second = fixture.rows[1]!; - expect(replay(first, command(first).replace('head -40', 'head -25'))).toEqual({ sourceRead: true, testsRead: true }); - expect(replay(second, command(second).replace('log --oneline -3', 'log --oneline -12'))).toEqual({ sourceRead: true, testsRead: true }); - }); - - test.each([ - ['awk action', (s: string) => s.replace("awk '/^### 3\\. Test review/,/^### 4\\./'", "awk 'BEGIN { system(\"cat fake\") }'")], - ['awk output redirection', (s: string) => s.replace("awk '/^### 3\\. Test review/,/^### 4\\./'", "awk '/x/ { print > \"src/billing.ts\" }'")], - ['shell substitution', (s: string) => s.replace('grep -n', 'grep -n "$(cat fake)"')], - ['backtick execution', (s: string) => s.replace('grep -n', 'grep -n `cat fake`')], - ['quoted injected command', (s: string) => s.replace('grep -n', 'grep -n "x"; printf fake; grep -n')], - ['read hidden in a conditional', (s: string) => s.replace('cat -n src/billing.ts', 'false && cat -n src/billing.ts')], - ['source-only filename', (s: string) => s.replace('cat -n src/billing.ts', "echo 'cat -n src/billing.ts'")], - ] as const)('%s cannot borrow source read evidence', (_, mutate) => { - const row = fixture.rows[0]!; - expect(mutate(command(row))).not.toBe(command(row)); - expect(replay(row, mutate(command(row))).sourceRead).toBe(false); - }); - - test.each([ - 'git log --output=src/billing.ts -3', - 'git log --ext-diff -3', - 'git log --format=%x00 -3', - 'git log -3; printf fake', - ])('unsupported Git command %s cannot borrow delivery', git => { - const row = fixture.rows[1]!; - expect(replay(row, command(row).replace('git log --oneline -3', git)).sourceRead).toBe(false); - }); - - test.each(['-f/tmp/other.awk', "'-f/tmp/other.awk'", "'--source=BEGIN {print \"fake\"}'"])( - 'awk input %s cannot introduce another program', operand => { - const row = fixture.rows[0]!; - const changed = command(row).replace("Test review/,/^### 4\\./' plan-eng-review/sections/review-sections.md", "Test review/,/^### 4\\./' " + operand); - expect(changed).not.toBe(command(row)); - expect(replay(row, changed).sourceRead).toBe(false); - }); - - test('successful command identity still requires the complete file and paired parent result', () => { - for (const row of fixture.rows) { - const missing = structuredClone(row) as any; - missing.transcript[2].message.content[0].content = 'src/billing.ts and test/billing.test.ts were read'; - expect(replay(missing)).toEqual({ sourceRead: false, testsRead: false }); - const failed = structuredClone(row) as any; - failed.transcript[2].message.content[0].is_error = true; - expect(replay(failed)).toEqual({ sourceRead: false, testsRead: false }); - } - }); -}); diff --git a/test/design-count-native-8525.test.ts b/test/design-count-native-8525.test.ts deleted file mode 100644 index 118ae791e..000000000 --- a/test/design-count-native-8525.test.ts +++ /dev/null @@ -1,137 +0,0 @@ -import { expect, test } from 'bun:test'; -import * as fs from 'node:fs'; -import * as os from 'node:os'; -import * as path from 'node:path'; -import fixture from './fixtures/design-count-native-8525.json'; -import { hasNativePlanTerminal } from './helpers/claude-pty-runner'; -import type { PlanCountTranscript } from './helpers/plan-count-transcript'; - -function completion() { - const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'design-8525-replay-')); - const file = path.join(dir, path.basename(fixture.provenance.planPath)); - const transcript = structuredClone(fixture.transcript) as PlanCountTranscript; - const edit = fixture.provenance.operations.filter(o => o.tool === 'Edit').at(-1)!; - const mtime = Date.parse(edit.acknowledgedAt) / 1000; - const write = (content = fixture.report) => { fs.writeFileSync(file, content); fs.utimesSync(file, mtime, mtime); }; - write(); - // The replay starts before the first retained native assistant message. - const startedAt = Math.min(...transcript.assistantMessages.map(m => Date.parse(m.timestamp))) - 1_000; - const final = transcript.assistantMessages.at(-1)!; - const check = () => hasNativePlanTerminal(transcript, file, startedAt, 'completion_summary'); - return { dir, file, transcript, final, write, check, cleanup: () => fs.rmSync(dir, {recursive:true, force:true}) }; -} - -test('exact native final text and reconstructed read-back-verified report supply completion', () => { - const f = completion(); try { expect(f.check()).toBe(true); } finally { f.cleanup(); } -}); -test('current typed status accepts presentation, field order and current report prose independently', () => { - const f=completion();try { - for (const heading of ['## Completion','### Completion summary','## Review complete','## Design review complete','**Review completion:**']) { - for (const status of ['STATUS: DONE','**STATUS:** DONE — review saved and verified.','**STATUS: DONE**']) { - for (const fields of [ - [status,`What changed: \`${path.basename(f.file)}\` now carries the current review report.`], - [`Report: ${f.file} contains the reviewed plan and verification.`,status], - [status,`- Plan saved to \`${f.file}\`.`], - ]) {f.final.text=heading+'\n\n'+fields.join('\n\n');expect(f.check(),f.final.text).toBe(true);} - } - } - } finally {f.cleanup();} -}); -for (const [name, change] of Object.entries({ - 'blocked':(s:string)=>s.replace('DONE —','BLOCKED —'), - 'concerns':(s:string)=>s.replace('DONE —','DONE_WITH_CONCERNS —'), - 'pending':(s:string)=>s.replace('DONE —','NEEDS_CONTEXT —'), - 'conditional status':(s:string)=>s.replace('DONE —','DONE if approved —'), - 'conditional reason':(s:string)=>s.replace('completed with evidence','will be completed with evidence'), - 'quoted status':(s:string)=>s.replace('**STATUS:**','> **STATUS:**'), - 'literal status':(s:string)=>s.replace(/\*\*STATUS:\*\* (.+)/,'`STATUS: $1`'), - 'fenced status':(s:string)=>s.replace(/\*\*STATUS:\*\* (.+)/,'```text\nSTATUS: $1\n```'), - 'duplicate status':(s:string)=>s+'\nSTATUS: DONE', - 'conflicting status':(s:string)=>s+'\nSTATUS: BLOCKED', - 'historical context':(s:string)=>'Previous result:\n\n'+s, - 'copied section':(s:string)=>'Source example:\n\n'+s, - 'quoted section':(s:string)=>'> '+s.replaceAll('\n','\n> '), - 'duplicate section':(s:string)=>s+'\n## Review complete\nSTATUS: DONE', - 'unavailable report':(s:string)=>s.replace('now carries','is unavailable; would contain'), - 'proposed write':(s:string)=>s.replace('now carries','will contain'), - 'historical report':(s:string)=>s.replace('now carries','previously contained'), - 'wrong path':(s:string)=>s.replaceAll('gstack-test-plan-design.md','wrong-plan.md'), - 'ambiguous path':(s:string)=>s.replace('now carries','and `another-plan.md` now carry'), - 'different absolute directory':(s:string)=>s.replaceAll('gstack-test-plan-design.md','/elsewhere/gstack-test-plan-design.md'), - 'relative traversal':(s:string)=>s.replaceAll('gstack-test-plan-design.md','../gstack-test-plan-design.md'), - 'quoted artifact line':(s:string)=>s.replace('**What changed:**','> **What changed:**'), - 'literal artifact prose':(s:string)=>s.replace(/\*\*What changed:\*\* (.+)/,'**What changed:** "$1"'), - 'missing artifact field':(s:string)=>s.replace(/^\*\*What changed:\*\*.+\n/m,''), - 'withdrawn report':(s:string)=>s+'\nThe report is withdrawn.', - 'remaining decision':(s:string)=>s+'\nOne design decision is unresolved.', -})) test(`typed delivery rejects ${name}`, () => {const f=completion();try {f.final.text=change(f.final.text);expect(f.check()).toBe(false);}finally{f.cleanup();}}); -test('typed delivery retains source session, answer chronology, fresh file and complete Design report checks', () => { - const f=completion();try { - const original=structuredClone(f.transcript); - for (const change of [ - (t:PlanCountTranscript)=>{t.calls[1]!.answered=false;}, - (t:PlanCountTranscript)=>{t.calls[1]!.failed=true;}, - (t:PlanCountTranscript)=>{t.calls[1]!.sessionId='foreign';}, - (t:PlanCountTranscript)=>{t.calls[1]!.answeredAt=t.assistantMessages.at(-1)!.timestamp;}, - (t:PlanCountTranscript)=>{t.assistantMessages.at(-1)!.timestamp='2999-01-01T00:00:00Z';}, - ]) {Object.assign(f.transcript,structuredClone(original));change(f.transcript);expect(f.check()).toBe(false);} - Object.assign(f.transcript,structuredClone(original)); - for (const body of ['# Draft',fixture.report+'\n## Implementation changes\n',fixture.report.replace('| 1 | clean |','| 1 | pending |'),fixture.report.replace('DESIGN CLEARED','NOT CLEARED'),fixture.report.replace('NO UNRESOLVED DECISIONS','**UNRESOLVED DECISIONS:**\n- Still open')]) {f.write(body);expect(f.check()).toBe(false);} - f.write();fs.utimesSync(f.file,1,1);expect(f.check()).toBe(false); - fs.rmSync(f.file);expect(f.check()).toBe(false); - const alternate=path.join(f.dir,'alternate.md');fs.writeFileSync(alternate,fixture.report);fs.symlinkSync(alternate,f.file);expect(f.check()).toBe(false); - }finally{f.cleanup();} -}); -const cf74 = fixture.cf74Retry; - -function cf74Completion() { - const dir=fs.mkdtempSync(path.join(os.tmpdir(),'design-cf74-completion-')); - const file=path.join(dir,path.basename(cf74.provenance.planPath)); - const transcript=structuredClone(cf74.transcript) as PlanCountTranscript; - const final=transcript.assistantMessages.at(-1)!; - final.text=final.text.replaceAll(cf74.provenance.planPath,file); - const startedAt=Math.min(...transcript.calls.map(c=>Date.parse(c.answeredAt!)))-1000; - const write=(body=cf74.report)=>{fs.writeFileSync(file,body);fs.utimesSync(file,cf74.provenance.reportMtimeMs/1000,cf74.provenance.reportMtimeMs/1000);}; - write(); - return {dir,file,transcript,final,startedAt,write,check:()=>hasNativePlanTerminal(transcript,file,startedAt,'completion_summary'),cleanup:()=>fs.rmSync(dir,{recursive:true,force:true})}; -} - -test('cf74 actual completed native report envelope binds the fresh owned Design report',()=>{ - const f=cf74Completion();try{expect(f.check()).toBe(true);}finally{f.cleanup();} -}); -for(const heading of ['## Completion report','### Completion summary','## Completion'])for(const field of ['Plan written:','Plan saved:','Plan written to']) - test(`cf74 complete typed delivery: ${heading}/${field}`,()=>{ - const f=cf74Completion();try{f.final.text=f.final.text.replace('## Completion report',heading).replace('Plan written:',field);expect(f.check()).toBe(true);}finally{f.cleanup();} - }); -for(const [name,change]of Object.entries({ - 'pending status':(s:string)=>s.replace('STATUS: DONE','STATUS: PENDING'), - 'conditional status':(s:string)=>s.replace('STATUS: DONE','STATUS: DONE if approved'), - 'quoted status':(s:string)=>s.replace('**STATUS: DONE**','`STATUS: DONE`'), - 'duplicate status':(s:string)=>s+'\nSTATUS: DONE', - 'quoted whole report':(s:string)=>'> '+s.replaceAll('\n','\n> '), - 'historical report':(s:string)=>s.replace('## Completion report','Historical source:\n\n## Completion report'), - 'duplicate report':(s:string)=>s+'\n## Completion report\nSTATUS: DONE', - 'future write':(s:string)=>s.replace('Plan written:','Plan will be written:'), - 'conditional write':(s:string)=>s.replace('Plan written:', 'Plan written if approved:'), - 'quoted written field':(s:string)=>s.replace('- **Plan written:**','> **Plan written:**'), - 'ambiguous path':(s:string)=>s.replace(' — accepted behavior',' and another-report.md — accepted behavior'), - 'foreign path':(s:string)=>s.replaceAll('gstack-test-plan-design.md','foreign-report.md'), - 'withdrawn report':(s:string)=>s+'\nThe review report is withdrawn.', - 'unresolved decision':(s:string)=>s+'\nOne design decision is unresolved.', - 'quoted current unresolved status':(s:string)=>s+'\nOne design decision is "unresolved".', -}))test(`cf74 typed completion rejects ${name}`,()=>{const f=cf74Completion();try{f.final.text=change(f.final.text);expect(f.check()).toBe(false);}finally{f.cleanup();}}); -test('cf74 typed envelope cannot bypass fresh own Design report and native chronology',()=>{ - const f=cf74Completion();try{ - const base=structuredClone(f.transcript); - for(const change of [ - (t:PlanCountTranscript)=>{t.calls[0]!.answered=false;},(t:PlanCountTranscript)=>{t.calls[0]!.failed=true;}, - (t:PlanCountTranscript)=>{t.calls[0]!.answers={};},(t:PlanCountTranscript)=>{t.calls[0]!.unansweredQuestionIndices=[0];}, - (t:PlanCountTranscript)=>{t.calls[0]!.sessionId='foreign';},(t:PlanCountTranscript)=>{t.calls[0]!.answeredAt=t.assistantMessages.at(-1)!.timestamp;}, - ]){Object.assign(f.transcript,structuredClone(base));change(f.transcript);expect(f.check()).toBe(false);} - Object.assign(f.transcript,structuredClone(base)); - for(const report of [cf74.report.replace('| 1 | clean |','| 1 | pending |'),cf74.report.replace('DESIGN CLEARED','DESIGN NOT CLEARED'),cf74.report.replace('NO UNRESOLVED DECISIONS','**UNRESOLVED DECISIONS:**\n- One pending'),cf74.report+'\n## Another section\n', '# Draft']){f.write(report);expect(f.check()).toBe(false);} - f.write();fs.utimesSync(f.file,1,1);expect(f.check()).toBe(false); - fs.rmSync(f.file);expect(f.check()).toBe(false); - const target=path.join(f.dir,'other.md');fs.writeFileSync(target,cf74.report);fs.symlinkSync(target,f.file);expect(f.check()).toBe(false); - }finally{f.cleanup();} -}); diff --git a/test/design-crop-gutter-ap.test.ts b/test/design-crop-gutter-ap.test.ts deleted file mode 100644 index 4c4509ec7..000000000 --- a/test/design-crop-gutter-ap.test.ts +++ /dev/null @@ -1,102 +0,0 @@ -import {expect,test} from 'bun:test'; -import fs from 'node:fs'; -import os from 'node:os'; -import path from 'node:path'; -import fixture from './fixtures/design-crop-gutter-ap.json'; -import previous from './fixtures/plan-count-crop-ak.json'; -import {currentFilePermissionEpoch} from './helpers/plan-count-file-permission'; -import {createPlanCountPermissionGuard} from './helpers/claude-pty-runner'; -import {E2E_TOUCHFILES,selectTests} from './helpers/touchfiles'; - -function replay(change:(f:any)=>void=()=>{},input:any=fixture){ - const dir=fs.mkdtempSync(path.join(os.tmpdir(),'design-gutter-ap-')); - const expected=path.join(dir,'report.md'),record=path.join(dir,'record.json'); - const f:any={expected,record,cwd:input.cwd,config:input.config,startedAt:input.startedAt, - screen:input.screen.replaceAll(path.dirname(input.hook.expected),dir).replaceAll(path.basename(input.hook.expected),'report.md'), - state:{...structuredClone(input.hook),expected},before:input.ownedBefore,transcript:structuredClone(input.transcript),fileKind:'file'}; - try{ - change(f);fs.writeFileSync(record,JSON.stringify(f.state)); - if(f.fileKind==='file')fs.writeFileSync(expected,f.before); - if(f.fileKind==='directory')fs.mkdirSync(expected); - if(f.fileKind==='symlink'){const target=path.join(dir,'other.md');fs.writeFileSync(target,f.before);fs.symlinkSync(target,expected);} - const epoch=currentFilePermissionEpoch(record,f.expected,f.cwd,f.config,f.startedAt,f.transcript,f.screen); - const guard=createPlanCountPermissionGuard(); - return {epoch,first:guard(f.screen,'',epoch),second:guard(f.screen,'',epoch)}; - }finally{fs.rmSync(dir,{recursive:true,force:true});} -} - -test('exact five-column current gutter supplies one owned Edit epoch and one grant',()=>{ - const lines=fixture.screen.split('\n');expect(lines[0]).toMatch(/^ {5}\S/);expect(lines[1]).toBe(' 62 '); - expect(fixture.ownedBefore.split('\n')[60]!.endsWith(lines[0]!.slice(5))).toBe(true); - const r=replay();expect(r.epoch).toEqual({pendingId:fixture.hook.pendingId,completedId:fixture.hook.completedId,completedIds:fixture.hook.completedIds}); - expect(r.first).toBe('grant');expect(r.second).toBe('handled'); -}); - -test('prior six-column public crop remains exact and one-time',()=>{ - expect(previous.screen.split('\n')[0]).toMatch(/^ {6}\S/);expect(previous.screen.split('\n')[1]).toBe(' 82 '); - const r=replay(()=>{},previous);expect(r.epoch?.pendingId).toBe(previous.hook.pendingId);expect(r.first).toBe('grant');expect(r.second).toBe('handled'); - // Keep the existing six-space acceptance even when the adjacent numeric - // row has a different padding; the unchanged original-line guard remains. - expect(replay(f=>{f.screen=' '+f.screen;}).epoch?.pendingId).toBe(fixture.hook.pendingId); -}); - -test('padding and line-number width derive the continuation column together',()=>{ - const padded=replay(f=>{f.screen=' '+f.screen;f.screen=f.screen.replace(/^ 62 $/m,' 62 ');}); - expect(padded.epoch?.pendingId).toBe(fixture.hook.pendingId); - const relocated=replay(f=>{ - f.before='Earlier unchanged line\n'.repeat(38)+f.before; - f.screen=' '+f.screen; - f.screen=f.screen.replace(/^ ([1-9]\d*)( | [+-])/gm,(_:string,n:string,g:string)=>' '+(Number(n)+38)+g); - }); - expect(relocated.epoch?.pendingId).toBe(fixture.hook.pendingId); -}); - -test.each([0,3])('a native numbered-row padding of %d derives a matching non-six gutter',padding=>{ - const r=replay(f=>{ - f.screen=' '.repeat(padding+4)+f.screen.slice(5); - f.screen=f.screen.replace(/^ 62 $/m,' '.repeat(padding)+'62 '); - }); - expect(r.epoch?.pendingId).toBe(fixture.hook.pendingId);expect(r.first).toBe('grant'); -}); - -test('an ordinary numbered unchanged row remains a numbered row, not a wrapped continuation',()=>{ - const r=replay(f=>{ - f.screen=' 61 '+f.before.split('\n')[60]+'\n'+f.screen.slice(f.screen.indexOf('\n')+1); - }); - expect(r.epoch?.pendingId).toBe(fixture.hook.pendingId);expect(r.first).toBe('grant'); -}); - -const negatives:Array<[string,(f:any)=>void]>=[ - ['four-space gutter with five-column numbered row',f=>{f.screen=f.screen.slice(1);}], - ['seven-space gutter with five-column numbered row',f=>{f.screen=' '+f.screen;}], - ['tab cannot substitute for a native space gutter',f=>{f.screen='\t'+f.screen.slice(1);}], - ['wrong preceding file line',f=>{f.screen=f.screen.replace(/^ 62 $/m,' 63 ');}], - ['changed continuation content',f=>{f.screen=f.screen.replace('f2 with icon','foreign with icon');}], - ['stale before-file bytes',f=>{f.before=f.before.replace('f2 with icon','changed with icon');}], - ['quoted continuation',f=>{f.screen=f.screen.replace(/^ {5}/,' > ');}], - ['two unnumbered continuation rows',f=>{f.screen=f.screen.split('\n')[0]+'\n'+f.screen;}], - ['next row is an addition, not unchanged context',f=>{f.screen=f.screen.replace(/^ 62 $/m,' 62 +');}], - ['zero next line',f=>{f.screen=f.screen.replace(/^ 62 $/m,' 00 ');}], - ['missing current file',f=>{f.fileKind='missing';}], - ['directory instead of current file',f=>{f.fileKind='directory';}], - ['required source line beyond the bounded prefix',f=>{f.before='x'.repeat(65537)+f.before;}], - ['foreign displayed directory',f=>{f.screen=f.screen.replace(path.dirname(f.expected)+' for this session',path.join(path.dirname(f.expected),'foreign')+' for this session');}], - ['foreign hook target',f=>{f.state.expected+='.foreign';}], - ['foreign hook cwd',f=>{f.state.cwd+='.foreign';}], - ['foreign native session',f=>{f.transcript={status:'ready',calls:[],assistantMessages:[{sessionId:'foreign',text:'Current review',timestamp:new Date(f.startedAt).toISOString()}]};}], - ['missing pending request',f=>{f.state.pendingId=null;}], - ['completed request cannot reopen',f=>{f.state.completedId=f.state.pendingId;}], - ['stale request timestamp',f=>{f.state.timestamp=new Date(f.startedAt-1).toISOString();}], - ['missing menu footer',f=>{f.screen=f.screen.replace('Esc to cancel · Tab to amend','');}], - ['one-time action changed',f=>{f.screen=f.screen.replace('❯ 1. Yes','❯ 1. Yes, always allow');}], -]; -test.each(negatives)('%s cannot obtain a grant',(_,change)=>{ - const r=replay(change);expect(r.epoch).not.toBeTruthy();expect(r.first).not.toBe('grant'); -}); -test.skipIf(process.platform==='win32')('symlink cannot provide the original line',()=>{expect(replay(f=>{f.fileKind='symlink';}).epoch).toBeNull();}); - -test('new public regression dependencies select exactly the existing file-permission owners',()=>{ - for(const file of ['test/design-crop-gutter-ap.test.ts','test/fixtures/design-crop-gutter-ap.json']){ - expect(selectTests([file],E2E_TOUCHFILES).selected.sort()).toEqual(selectTests(['test/helpers/plan-count-file-permission.ts'],E2E_TOUCHFILES).selected.sort()); - } -}); diff --git a/test/design-scope-announcement-ao.test.ts b/test/design-scope-announcement-ao.test.ts deleted file mode 100644 index 07a03025f..000000000 --- a/test/design-scope-announcement-ao.test.ts +++ /dev/null @@ -1,80 +0,0 @@ -import { expect, test } from 'bun:test'; -import captured from './fixtures/design-scope-announcement-ao.json'; -import { nativeSeededPlanSelection } from './helpers/plan-scope-selection'; -import { E2E_TOUCHFILES, selectTests } from './helpers/touchfiles'; -import type { PlanCountTranscript, NativePublicToolEvent } from './helpers/plan-count-transcript'; - -const input = () => structuredClone(captured.projection); -type Input = ReturnType; -const announcement = (p: Input) => p.transcript.assistantMessages.find(m => m.sessionId === p.opts.sessionId && m.text.startsWith("I'll auto-select"))!; -const verdict = (p: Input) => nativeSeededPlanSelection(p.transcript as PlanCountTranscript, p.tools as NativePublicToolEvent[], p.opts); - -test('the exact owned post-load option B announcement selects the seeded title', () => { - expect(captured.rawScopeGateAutoSelectObserved).toBe(false); - expect(verdict(input())).toBe(true); -}); - -test('equivalent explicit selection words and balanced title quotes retain identity', () => { - for (const prefix of ["I'll auto-select", 'I will auto-select', "I'll auto select"]) { - for (const title of ['Marketing landing page', '"Marketing landing page"', '“Marketing landing page”', '`Marketing landing page`']) { - const p = input(), m = announcement(p); - m.text = m.text.replace("I'll auto-select", prefix).replace('Marketing landing page', title); - expect(verdict(p)).toBe(true); - } - } - const p = input(), m = announcement(p); - p.opts.seed = p.opts.seed.replace('Marketing landing page', 'Account settings'); - m.text = m.text.replace('Marketing landing page', 'Account settings'); - expect(verdict(p)).toBe(true); -}); - -const rejected: Array<[string, (p: Input) => void]> = [ - ['wrong option', p => { announcement(p).text = announcement(p).text.replace('option B', 'option A'); }], - ['wrong target', p => { announcement(p).text = announcement(p).text.replace('Marketing landing page', 'Account settings'); }], - ['target prefix only', p => { announcement(p).text = announcement(p).text.replace('page draft', 'page experiment draft'); }], - ['conditional selection', p => { announcement(p).text = 'If approved: ' + announcement(p).text; }], - ['source selection', p => { announcement(p).text = 'Source excerpt:\n' + announcement(p).text; }], - ['quoted selection', p => { announcement(p).text = '> ' + announcement(p).text; }], - ['wholly quoted selection', p => { announcement(p).text = '"' + announcement(p).text + '"'; }], - ['unbalanced target quotes', p => { announcement(p).text = announcement(p).text.replace('Marketing landing page', '"Marketing landing page'); }], - ['question instead of assertion', p => { announcement(p).text = announcement(p).text.replace(/\.$/, '?'); }], - ['conditional tail', p => { announcement(p).text = announcement(p).text.replace(', running', ' if approved, running'); }], - ['cancelled selection', p => { announcement(p).text += '\nCorrection: this selection is withdrawn.'; }], - ['quoted status cancellation', p => { announcement(p).text += '\nThis selection is "withdrawn".'; }], - ['replaced target', p => { announcement(p).text += '\nThe selected target is now the branch diff.'; }], - ['pre-invocation announcement', p => { announcement(p).timestamp = new Date(p.opts.commandStartedAt - 1).toISOString(); }], - ['foreign announcement', p => { announcement(p).sessionId = 'foreign'; }], - ['foreign load result', p => { p.tools[1]!.sessionId = 'foreign'; }], - ['failed skill load', p => { p.tools[1]!.isError = true; }], - ['wrong skill', p => { p.tools[0]!.input!.skill = 'plan-eng-review'; }], - ['late command start', p => { p.opts.commandStartedAt = Date.parse(p.tools[1]!.timestamp) + 1; }], - ['multiple seed titles', p => { p.opts.seed += '\n# Another plan\n'; }], -]; -test.each(rejected)('%s supplies no scope selection', (_, change) => { - const p = input(); p.transcript.assistantMessages = [announcement(p)]; change(p); expect(verdict(p)).toBe(false); -}); - -test('quoted historical or foreign withdrawals do not replace the current selection', () => { - for (const correction of ['> This selection is withdrawn.', 'Historical note: "This selection is withdrawn."']) { - const p = input(); announcement(p).text += '\n' + correction; expect(verdict(p)).toBe(true); - } - const p = input(), m = announcement(p); - p.transcript.assistantMessages.push({ ...m, sessionId: 'foreign', text: 'This selection is withdrawn.' }); - expect(verdict(p)).toBe(true); -}); - -test('a later current withdrawal invalidates selection until a later reselection', () => { - const p = input(), m = announcement(p); - p.transcript.assistantMessages.push({ ...m, timestamp: new Date(Date.parse(m.timestamp) + 1000).toISOString(), text: 'This selection is withdrawn.' }); - expect(verdict(p)).toBe(false); - p.transcript.assistantMessages.push({ ...m, timestamp: new Date(Date.parse(m.timestamp) + 2000).toISOString() }); - expect(verdict(p)).toBe(true); -}); - -test('both regression sources select the same five existing scope observers', () => { - const expected = selectTests(['test/helpers/plan-scope-selection.ts'], E2E_TOUCHFILES, []).selected; - expect(expected).toHaveLength(5); - for (const path of ['test/design-scope-announcement-ao.test.ts', 'test/fixtures/design-scope-announcement-ao.json']) { - expect(selectTests([path], E2E_TOUCHFILES, []).selected).toEqual(expected); - } -}); diff --git a/test/design-scope-declaration-ak.test.ts b/test/design-scope-declaration-ak.test.ts deleted file mode 100644 index d6aa75cc2..000000000 --- a/test/design-scope-declaration-ak.test.ts +++ /dev/null @@ -1,91 +0,0 @@ -import { expect, test } from 'bun:test'; -import { nativeSeededPlanSelection } from './helpers/plan-scope-selection'; -import fixture from './fixtures/design-scope-declaration-ak.json'; -import type { PlanCountTranscript, NativePublicToolEvent } from './helpers/plan-count-transcript'; -import { E2E_TOUCHFILES, selectTests } from './helpers/touchfiles'; - -const input = (attempt = 0) => structuredClone(fixture.attempts[attempt]!.projection); -const verdict = (p = input()) => nativeSeededPlanSelection(p.transcript as PlanCountTranscript, p.tools as NativePublicToolEvent[], p.opts); -const declaration = (p: ReturnType) => p.transcript.assistantMessages.find(m => /^(?:I'll proceed with reviewing|Scope gate confirms plan mode)/.test(m.text))!; - -test('both exact owned post-load announcements select the named pasted draft', () => { - for (let attempt = 0; attempt < 2; attempt++) { - const p = input(attempt); - expect(fixture.attempts[attempt]!.rawScopeGateAutoSelectObserved).toBe(false); - expect(verdict(p)).toBe(true); - } -}); - -test('the prior AJ fresh unique-draft introduction now binds without changing its recorded outcome', () => { - const p = fixture.priorGenuineFailure.projection; - expect(nativeSeededPlanSelection(p.transcript as PlanCountTranscript, p.tools as NativePublicToolEvent[], p.opts)).toBe(true); -}); - -test('target identity and ordinary equivalent current review wording remain bound', () => { - for (let attempt = 0; attempt < 2; attempt++) { - const p = input(attempt); p.opts.seed = p.opts.seed.replace('Marketing landing page', 'Account settings'); - p.transcript.assistantMessages.forEach(m => { m.text = m.text.replaceAll('Marketing landing page', 'Account settings'); }); - for (const t of p.tools) if (t.input?.args) t.input.args = t.input.args.replaceAll('Marketing landing page', 'Account settings'); - expect(verdict(p)).toBe(true); - } - const p = input(); declaration(p).text = declaration(p).text.replace("I'll proceed", 'I will proceed'); expect(verdict(p)).toBe(true); -}); - -test('source, historical, quoted, hypothetical and conditional introductions do not select', () => { - for (let attempt = 0; attempt < 2; attempt++) for (const prefix of [ - '> ', ' ', 'Source excerpt:\n', 'Historical example only.\n', 'The following is hypothetical. ', 'If approved, ', '```\n', '"', - ]) { - const p = input(attempt), m = declaration(p); p.transcript.assistantMessages = [m]; m.text = prefix + m.text; - expect(verdict(p)).toBe(false); - } -}); - -test('a different target or conditional scope announcement cannot borrow the draft name', () => { - for (let attempt = 0; attempt < 2; attempt++) for (const change of [ - (s: string) => s.replaceAll('Marketing landing page', 'Checkout redesign'), - (s: string) => s.replace(/draft(?: plan)?/, 'draft plan if approved'), - ]) { const p = input(attempt), m = declaration(p); p.transcript.assistantMessages = [m]; m.text = change(m.text); expect(verdict(p)).toBe(false); } - for (const prefix of ['Scope gate might confirm plan mode, so', 'Scope gate confirms branch mode, so']) { - const p = input(1); p.transcript.assistantMessages = [declaration(p)]; declaration(p).text = declaration(p).text.replace('Scope gate confirms plan mode, so', prefix); expect(verdict(p)).toBe(false); - } -}); - -test('the same successful Skill load and post-command current session remain necessary', () => { - for (let attempt = 0; attempt < 2; attempt++) for (const change of [ - (p: ReturnType) => { p.opts.sessionId = 'foreign'; }, - (p: ReturnType) => { p.tools[1]!.isError = true; }, - (p: ReturnType) => { p.tools[1]!.toolUseId = 'foreign'; }, - (p: ReturnType) => { p.tools[0]!.input!.skill = 'plan-eng-review'; }, - (p: ReturnType) => { p.opts.commandStartedAt = Date.parse(p.tools[1]!.timestamp) + 1; }, - (p: ReturnType) => { declaration(p).timestamp = new Date(p.opts.commandStartedAt - 1).toISOString(); p.transcript.assistantMessages = [declaration(p)]; }, - ]) { const p = input(attempt); change(p); expect(verdict(p)).toBe(false); } -}); - -test('same-message and later current withdrawals or replacement targets defeat selection', () => { - for (let attempt = 0; attempt < 2; attempt++) for (const correction of [ - 'Correction: this selection is withdrawn.', - 'The selected target is now the branch diff.', - 'This declaration has been retracted.', - ]) for (const later of [false, true]) { - const p = input(attempt), m = declaration(p); p.transcript.assistantMessages = [m]; - if (later) p.transcript.assistantMessages.push({ ...m, timestamp: new Date(Date.parse(m.timestamp) + 1000).toISOString(), text: correction }); - else m.text += '\n' + correction; - expect(verdict(p)).toBe(false); - } -}); - -test('literal or foreign corrections preserve the actual declaration and a later reselection is current', () => { - for (let attempt = 0; attempt < 2; attempt++) { - const p = input(attempt), m = declaration(p); p.transcript.assistantMessages = [m]; - p.transcript.assistantMessages.push({ ...m, sessionId: 'foreign', text: 'The selected target is now the branch diff.' }); - p.transcript.assistantMessages.push({ ...m, text: '> This selection is withdrawn.' }); expect(verdict(p)).toBe(true); - p.transcript.assistantMessages.push({ ...m, timestamp: new Date(Date.parse(m.timestamp) + 1000).toISOString(), text: 'This selection is withdrawn.' }); expect(verdict(p)).toBe(false); - p.transcript.assistantMessages.push({ ...m, timestamp: new Date(Date.parse(m.timestamp) + 2000).toISOString() }); expect(verdict(p)).toBe(true); - } -}); - -test('the new evidence dependencies select exactly the existing five scope observers', () => { - const expected = selectTests(['test/helpers/plan-scope-selection.ts'], E2E_TOUCHFILES, []).selected; - expect(expected).toHaveLength(5); - for (const path of ['test/design-scope-declaration-ak.test.ts','test/fixtures/design-scope-declaration-ak.json']) expect(selectTests([path], E2E_TOUCHFILES, []).selected).toEqual(expected); -}); diff --git a/test/design-scope-entry-aq.test.ts b/test/design-scope-entry-aq.test.ts deleted file mode 100644 index fc05b3f13..000000000 --- a/test/design-scope-entry-aq.test.ts +++ /dev/null @@ -1,89 +0,0 @@ -import {expect, test} from 'bun:test'; -import fs from 'node:fs'; -import path from 'node:path'; -import {ALL_HOST_CONFIGS} from '../hosts'; -import {HOST_PATHS, type TemplateContext} from '../scripts/resolvers/types'; -import {generatePreamble} from '../scripts/resolvers/preamble'; -import {generateBaseBranchDetect} from '../scripts/resolvers/utility'; -import {E2E_TOUCHFILES, LLM_JUDGE_TOUCHFILES, selectTests} from './helpers/touchfiles'; -import failedScopes from './fixtures/design-scope-checkpoint-at.json'; -import {nativeSeededPlanSelection} from './helpers/plan-scope-selection'; -import {isScopeGateAutoSelectVisible} from './helpers/claude-pty-runner'; - -const template = fs.readFileSync(path.join(import.meta.dir, '../plan-design-review/SKILL.md.tmpl'), 'utf8'); -const scope = template.slice(template.indexOf('## Scope gate'), template.indexOf('## Design Philosophy')); -const announcement = 'Scope gate: plan mode — auto-selected B (reviewing ).'; - -test('Design resolves scope before either executable bootstrap placeholder', () => { - const gate = template.indexOf('## Scope gate'); - expect(gate).toBeGreaterThan(0); - for (const token of ['{{PREAMBLE}}', '{{BASE_BRANCH_DETECT}}']) { - expect(template.split(token)).toHaveLength(2); - expect(template.indexOf(announcement)).toBeLessThan(template.indexOf(token)); - expect(template.indexOf('Reply with A, B, or C. STOP and wait')).toBeLessThan(template.indexOf(token)); - } - expect(template.indexOf('{{PREAMBLE}}')).toBeLessThan(template.indexOf('{{BASE_BRANCH_DETECT}}')); - expect(template.indexOf('{{BASE_BRANCH_DETECT}}')).toBeLessThan(template.indexOf('## Design Philosophy')); -}); - -test('every host expands its real bootstrap after the mandatory entry gate', () => { - for (const host of ALL_HOST_CONFIGS) { - const ctx: TemplateContext = {skillName: 'plan-design-review', tmplPath: 'plan-design-review/SKILL.md.tmpl', - host: host.name, paths: HOST_PATHS[host.name]!, preambleTier: 3, interactive: true}; - const preamble = generatePreamble(ctx); - const brain = host.suppressedResolvers?.includes('BASE_BRANCH_DETECT') ? '' : generateBaseBranchDetect(ctx); - const expanded = template.replace('{{PREAMBLE}}', preamble).replace('{{BASE_BRANCH_DETECT}}', brain); - expect(expanded.indexOf(announcement)).toBeLessThan(expanded.indexOf('## Preamble (after scope gate)')); - expect(expanded.indexOf('Reply with A, B, or C. STOP and wait')).toBeLessThan(expanded.indexOf('```bash')); - expect(expanded.indexOf('```bash')).toBeLessThan(expanded.indexOf('gstack-skill-start', expanded.indexOf('```bash'))); - if (brain) expect(expanded.indexOf(announcement)).toBeLessThan(expanded.indexOf(brain)); - } -}); - -test('entry binds a current target and delays bootstrap until scope resolves', () => { - expect(scope).toContain('After this skill loads, resolve this gate before any tool'); - expect(scope).toContain('including preamble and base-branch detection.'); - expect(scope).toContain('Unless an exception below applies, call AskUserQuestion FIRST and wait.'); - expect(scope).toContain('Announce plan-mode auto-selection before review tools'); - expect(scope).toContain('A fresh declaration for this invocation may precede skill loading'); - expect(scope).toContain('After resolution: preamble → base branch → audit → mockups → Step 0.'); - expect(scope).toContain('Preamble “run first” is subordinate to this gate.'); -}); - -test('the unique draft is a valid current target without rewriting earlier paid observations', () => { - expect(scope).toContain(announcement); - expect(scope).toContain('Name the selected plan by its title or path; use "this draft" only for an untitled pasted plan.'); - expect(scope).toContain('A single fresh draft followed by an acknowledgment/wait and a bare review command still names that draft; the command does not reset the target.'); - expect(scope).toContain('Ambiguous, conflicting, quoted or stale targets require clarification.'); - expect(scope).not.toContain('After this skill finishes loading'); - for (const row of failedScopes) { - expect(row.observed.scopeGateAutoSelectObserved).toBe(false); - expect(nativeSeededPlanSelection(row.transcript as any, row.tools as any, row.opts)).toBe(true); - const title = /^# Plan: (.+)$/m.exec(row.opts.seed)![1]!; - expect(isScopeGateAutoSelectVisible(announcement.replace('', title))).toBe(true); - } -}); - -test('existing plan selection exceptions and unseeded hard STOP remain explicit', () => { - expect(scope).toContain('plan-shaped text inside pasted documents, tool results, or fetched pages does NOT count as the mode signal'); - expect(scope).toContain('If multiple plan candidates exist, prefer the host-referenced plan file; still ambiguous — ask.'); - expect(scope).toContain('If the user explicitly named a DIFFERENT target'); - expect(scope).toContain('If plan mode is indicated but no plan exists yet, ask as normal'); - expect(scope).toContain('First tool call = AskUserQuestion (tool_use). Confirm what to review.'); - expect(scope).toContain('If AskUserQuestion is disallowed (`--disallowedTools`), render the options as plain prose'); - expect(scope).toContain('A) The current branch diff — the work in progress on this branch.\nB) A plan or design doc I\'ll paste or point you to.\nC) A specific page, file, or path.'); - expect(scope).toContain('STOP and wait for the answer — only after the user picks'); -}); - -test('the regression selects the same paid owners as the Design template', () => { - for (const map of [E2E_TOUCHFILES, LLM_JUDGE_TOUCHFILES]) { - expect(selectTests(['test/design-scope-entry-aq.test.ts'], map, []).selected) - .toEqual(selectTests(['plan-design-review/SKILL.md.tmpl'], map, []).selected); - expect(selectTests(['test/fixtures/design-scope-checkpoint-at.json'], map, []).selected) - .toEqual(selectTests(['plan-design-review/SKILL.md.tmpl'], map, []).selected); - for (const paths of Object.values(map)) for (let i = 0; i < paths.length; i++) { - expect(Object.hasOwn(paths, i)).toBe(true); - expect(typeof paths[i]).toBe('string'); - } - } -}); diff --git a/test/design-scope-selection-aj.test.ts b/test/design-scope-selection-aj.test.ts deleted file mode 100644 index 8ff6b58aa..000000000 --- a/test/design-scope-selection-aj.test.ts +++ /dev/null @@ -1,110 +0,0 @@ -import { expect, test } from 'bun:test'; -import capture from './fixtures/design-scope-selection-aj.json'; -import { nativeSeededPlanSelection } from './helpers/plan-scope-selection'; -import type { PlanCountTranscript, NativePublicToolEvent } from './helpers/plan-count-transcript'; -import { selectTests, E2E_TOUCHFILES } from './helpers/touchfiles'; - -const originals = capture.observations; -const check = (observation = structuredClone(originals[0]!)) => nativeSeededPlanSelection( - observation.transcript as PlanCountTranscript, - observation.tools as NativePublicToolEvent[], - observation.opts, -); -const selectedMessage = (o: typeof originals[number]) => o.transcript.assistantMessages.find(m => m.text.includes('"Marketing landing page"'))!; - -test('both actual explicit draft selections bind the named seed after this session loaded the skill', () => { - for (const o of originals) expect(check(o)).toBe(true); - for (const verb of ["I'll review", 'I will review', "I'll go with reviewing", 'I will go with reviewing']) { - const o = structuredClone(originals[1]!); - selectedMessage(o).text = `${verb} the "Marketing landing page" draft, starting by checking the design system.`; - expect(check(o)).toBe(true); - } -}); - -test('a named target still requires the successful current skill and invocation', () => { - for (const original of originals) { - for (const mutate of [ - (o: typeof original) => { o.opts.seed = '# Plan: Other page'; }, - (o: typeof original) => { o.opts.seed += '\n# Plan: Another'; }, - (o: typeof original) => { o.opts.sessionId = 'foreign'; }, - (o: typeof original) => { o.opts.commandStartedAt = Date.parse(selectedMessage(o).timestamp) + 1; }, - (o: typeof original) => { o.tools[0]!.input!.skill = 'plan-ceo-review'; }, - (o: typeof original) => { o.tools[1]!.isError = true; }, - (o: typeof original) => { o.tools[1]!.toolUseId = 'foreign'; }, - (o: typeof original) => { o.tools.pop(); }, - ]) { - const o = structuredClone(original); o.transcript.assistantMessages = [selectedMessage(o)]; mutate(o); expect(check(o)).toBe(false); - } - } -}); - -test('quoted, hypothetical, conditional and withdrawn selections do not select the seed', () => { - for (const original of originals) { - const text = selectedMessage(original).text.trim(); - for (const invalid of [ - '> ' + text, ' ' + text, '"' + text + '"', 'Example:\n' + text, - 'The following is a source excerpt.\n' + text, 'An unproven hypothesis.\n' + text, - text.replace("I'll", 'I might'), text.replace("I'll", "I won't"), - text.replace('Marketing landing page', 'Other page'), - text.replace('draft', 'branch diff'), text.replace(/,$/, '?'), - text.replace(', ', ', if approved, '), - text + ' I retract that selection.', text + ' This selection is withdrawn.', - text + ' Treat that declaration as a hypothetical example.', - ].filter(value => value !== text)) { - const o = structuredClone(original); o.transcript.assistantMessages = [selectedMessage(o)]; selectedMessage(o).text = invalid; - expect(check(o), invalid).toBe(false); - } - } -}); - -test('scope selection remains mapped to the existing design and engineering mode workflows', () => { - for (const file of ['test/design-scope-selection-aj.test.ts', 'test/fixtures/design-scope-selection-aj.json']) { - expect(selectTests([file], E2E_TOUCHFILES, []).selected.sort()).toEqual(['plan-design-review-plan-mode', 'plan-eng-review-plan-mode']); - } -}); - -test('a complete owned observation cannot use a withdrawn selection or a replacement target', () => { - for (const original of originals) { - for (const correction of ['The selection has been withdrawn.', 'The selected target is now the branch diff.', 'I have withdrawn this selection.', 'Correction: The selected target is now the branch diff.']) { - for (const separator of [' ', '\n\n']) { - const o = structuredClone(original); - selectedMessage(o).text = selectedMessage(o).text.trim() + separator + correction; - expect(check(o)).toBe(false); - } - const o = structuredClone(original); - o.transcript.assistantMessages.push({ sessionId: o.opts.sessionId, timestamp: new Date(Date.parse(selectedMessage(o).timestamp) + 1000).toISOString(), text: correction }); - expect(check(o)).toBe(false); - } - } -}); - -test('old, unrelated, foreign and quoted assessments do not withdraw the current target', () => { - for (const original of originals) { - for (const text of [ - 'Old note: "The selection has been withdrawn."', - '> The selection has been withdrawn.', - '```text\nThe selected target is now the branch diff.\n```', - 'Source excerpt:\nThe selection has been withdrawn.', - 'The following is a hypothetical example.\nThe selected target is now the branch diff.', - 'An unrelated payment selection has been withdrawn.', - 'The selected target is now the "Marketing landing page" draft.', - 'If approved, the selection has been withdrawn.', - 'The selected target is now the branch diff?', - 'The selected target is now the branch diff? This is a question.', - 'I have withdrawn this selection?', - ]) { - const o = structuredClone(original); - o.transcript.assistantMessages.push({ sessionId: o.opts.sessionId, timestamp: new Date(Date.parse(selectedMessage(o).timestamp) + 1000).toISOString(), text }); - expect(check(o), text).toBe(true); - } - for (const foreign of [false, true]) { - const o = structuredClone(original); - o.transcript.assistantMessages.push({ sessionId: foreign ? 'foreign' : o.opts.sessionId, timestamp: new Date(Date.parse(selectedMessage(o).timestamp) + (foreign ? 1000 : -1000)).toISOString(), text: 'The selection has been withdrawn.' }); - expect(check(o)).toBe(true); - } - const o = structuredClone(original), selected = structuredClone(selectedMessage(o)); - o.transcript.assistantMessages.push({ sessionId: o.opts.sessionId, timestamp: new Date(Date.parse(selected.timestamp) + 1000).toISOString(), text: 'The selection has been withdrawn.' }); - o.transcript.assistantMessages.push({ ...selected, timestamp: new Date(Date.parse(selected.timestamp) + 2000).toISOString() }); - expect(check(o)).toBe(true); - } -}); diff --git a/test/dx-manual-handoff-ao.test.ts b/test/dx-manual-handoff-ao.test.ts deleted file mode 100644 index 98cbba67d..000000000 --- a/test/dx-manual-handoff-ao.test.ts +++ /dev/null @@ -1,103 +0,0 @@ -import {describe,expect,test} from 'bun:test'; -import fs from 'node:fs'; -import os from 'node:os'; -import path from 'node:path'; -import {hasNativePlanTerminal, classifyPlanCountFrame} from './helpers/claude-pty-runner'; -import type {NativePlanQuestionCall,PlanCountTranscript} from './helpers/plan-count-transcript'; -import captured from './fixtures/dx-manual-handoff-ao.json'; -import {E2E_TOUCHFILES,LLM_JUDGE_TOUCHFILES,GLOBAL_TOUCHFILES} from './helpers/touchfiles-data'; - -type Edit=(calls:NativePlanQuestionCall[], transcript:PlanCountTranscript, report:string)=>void; -function replay(edit?:Edit){ - const dir=fs.mkdtempSync(path.join(os.tmpdir(),'dx-manual-handoff-ao-')); - try { - const report=path.join(dir,'report.md');fs.writeFileSync(report,captured.reportContent); - const written=captured.provenance.reportMtimeMs/1000;fs.utimesSync(report,written,written); - const transcript={status:'ready',calls:structuredClone(captured.calls),assistantMessages:[],planReadyRequests:structuredClone(captured.planReadyRequests)} as PlanCountTranscript; - edit?.(transcript.calls,transcript,report); - return hasNativePlanTerminal(transcript,report,captured.provenance.startedAt,'plan_ready'); - } finally {fs.rmSync(dir,{recursive:true,force:true});} -} -function change(call:NativePlanQuestionCall,from:string,to:string){ - const q=call.questions[0]!;expect(q.question).toContain(from); - const selected=call.answers![q.question];q.question=q.question.replace(from,to);call.answers={[q.question]:selected!}; -} -describe('AO completed manual DX handoff preserves report freshness',()=>{ - test('shared completion callers register the regression with dense literal paths',()=>{ - const arrays=[...Object.values(E2E_TOUCHFILES),...Object.values(LLM_JUDGE_TOUCHFILES),GLOBAL_TOUCHFILES]; - expect(arrays).toHaveLength(216); - for(const values of arrays)for(let i=0;i{ - expect(captured.calls).toHaveLength(2);expect(captured.events).toHaveLength(4); - expect(Date.parse(captured.calls[0]!.answeredAt!)).toBeLessThan(captured.provenance.reportMtimeMs); - expect(Date.parse(captured.calls[1]!.answeredAt!)).toBeGreaterThan(captured.provenance.reportMtimeMs); - expect(classifyPlanCountFrame(captured.screen)).toBe('plan_ready'); - expect(replay()).toBe(true); - }); - test('equivalent completed recap and manual roles retain current authority',()=>{ - for(const [from,to] of [ - [' (5/10 -> 8.5/10)',''], - ['5/10 -> 8.5/10','6/10 → 9/10'], - ['What should happen next?',"What's next?"], - ['The DX review found','The DX review identified'], - ['All are written into the plan as tasks T1 to T9.','All DX decisions and tasks are recorded in the plan.'], - ])expect(replay(calls=>change(calls[1]!,from!,to!)),to).toBe(true); - expect(replay(calls=>calls[1]!.questions[0]!.options.reverse())).toBe(true); - expect(replay(calls=>{const c=calls[1]!;change(c,'Net: hand off now as you asked, or chain the eng review here.','Net: hand off now as you asked, or chain the eng review here.\n> Historical example: add a new task before leaving.');})).toBe(true); - expect(replay(calls=>{const o=calls[1]!.questions[0]!.options[0]!;o.description=o.description!.replace('Plan exits now with all DX decisions and tasks recorded; nothing else is started.','Exit the plan now with all DX tasks and decisions recorded. No further review is started.');})).toBe(true); - }); - test('source, conditional or withdrawn completion facts cannot make a stale report current',()=>{ - const edits:Array<[string,string]>=[ - ['D13 — DX review','Source: D13 — DX review'], - ['DX review complete','DX review is not complete'], - ['DX review complete','DX review complete only after another decision'], - ['ELI10: The DX review found','ELI10: Earlier review assessment: The DX review found'], - ['ELI10: The DX review found','ELI10: If approved, the DX review found'], - ['All are written into the plan as tasks T1 to T9.','Example: All are written into the plan as tasks T1 to T9.'], - ['All are written into the plan as tasks T1 to T9.','Previously, all are written into the plan as tasks T1 to T9.'], - ['All are written into the plan as tasks T1 to T9.','"All are written into the plan as tasks T1 to T9."'], - ['All are written into the plan as tasks T1 to T9.','All will be written into the plan as tasks T1 to T9.'], - ['Project/branch/task:','Source:\nProject/branch/task:'], - ['Project/branch/task: ','Project/branch/task: If approved, '], - ['Project/branch/task: ','Project/branch/task: Source excerpt, not a current assessment: '], - ]; - for(const [from,to] of edits)expect(replay(calls=>change(calls[1]!,from,to)),to).toBe(false); - for(const suffix of [' This review is not complete.',' These tasks are not recorded.',' This review is "withdrawn".',' One DX decision remains unresolved.',' We must fix another issue.',' Add another migration task.',' Should we approve another change?',' ']) { - expect(replay(calls=>{const c=calls[1]!;change(c,c.questions[0]!.question,c.questions[0]!.question+suffix);}),suffix).toBe(false); - } - }); - test('selected manual action must close with recorded decisions and no new work',()=>{ - for(const prefix of ['Source: ','Earlier review assessment: ','If approved, ','> '])expect(replay(calls=>{const o=calls[1]!.questions[0]!.options[0]!;o.description=prefix+o.description;}),prefix).toBe(false); - for(const suffix of [' Also update the plan before exit.',' Run /plan-eng-review now.',' This plan is not complete.',' The tasks are "withdrawn".',' This manual handoff is cancelled.'])expect(replay(calls=>{calls[1]!.questions[0]!.options[0]!.description+=suffix;}),suffix).toBe(false); - for(const from of ['all DX decisions and tasks recorded','nothing else is started'])expect(replay(calls=>{const o=calls[1]!.questions[0]!.options[0]!;o.description=o.description!.replace(from,'more work remains');}),from).toBe(false); - for(const index of [1,2])expect(replay(calls=>{const c=calls[1]!,q=c.questions[0]!;c.answers={[q.question]:q.options[index]!.label};})).toBe(false); - }); - test('completed owned native answer identity remains mandatory',()=>{ - const mutations:Array<(c:NativePlanQuestionCall)=>void>=[ - c=>{c.answered=false;},c=>{c.failed=true;},c=>{c.sessionId='foreign';},c=>{c.toolUseId='';}, - c=>{c.answeredAt='invalid';},c=>{c.answeredAt=new Date(Date.now()+60_000).toISOString();}, - c=>{c.unansweredQuestionIndices=[0];},c=>{delete c.unansweredQuestionIndices;}, - c=>{c.questions[0]!.multiSelect=true;},c=>{c.questions[0]!.header='Issue decision';}, - c=>{c.questions.push(structuredClone(c.questions[0]!));}, - c=>{c.answers={wrong:c.questions[0]!.options[0]!.label};}, - c=>{c.answers![c.questions[0]!.question]='Not an offered answer';}, - c=>{c.answers!.extra='foreign';}, - c=>{c.questions[0]!.options[1]!.label=c.questions[0]!.options[0]!.label;}, - ]; - for(const edit of mutations)expect(replay(calls=>edit(calls[1]!)),edit.toString()).toBe(false); - }); - test('other modifying answers and complete report/current Exit gates remain unchanged',()=>{ - expect(replay(calls=>{calls[0]!.answeredAt=new Date(captured.provenance.reportMtimeMs+1).toISOString();})).toBe(false); - expect(replay((_calls,_t,report)=>{const time=Date.parse(captured.calls[0]!.answeredAt!)/1000-1;fs.utimesSync(report,time,time);})).toBe(false); - expect(replay((_calls,_t,report)=>fs.writeFileSync(report,'# Completion summary\nDone.'))).toBe(false); - expect(replay((_calls,_t,report)=>fs.unlinkSync(report))).toBe(false); - for(const mutate of [ - (t:PlanCountTranscript)=>{t.planReadyRequests=[];}, - (t:PlanCountTranscript)=>{t.planReadyRequests![0]!.failed=true;}, - (t:PlanCountTranscript)=>{t.planReadyRequests![0]!.sessionId='foreign';}, - (t:PlanCountTranscript)=>{t.planReadyRequests![0]!.timestamp=captured.calls[1]!.answeredAt!;}, - (t:PlanCountTranscript)=>{t.status='missing';}, - ])expect(replay((_calls,t)=>mutate(t))).toBe(false); - }); -}); diff --git a/test/eng-annotated-cache-au.test.ts b/test/eng-annotated-cache-au.test.ts deleted file mode 100644 index 61fdd8f6a..000000000 --- a/test/eng-annotated-cache-au.test.ts +++ /dev/null @@ -1,57 +0,0 @@ -import {expect,test} from 'bun:test'; -import captured from './fixtures/eng-annotated-cache-au.json'; -import {engFirstReviewAUQ,engSetupAUQ,engStep0Boundary,nativePlanCallFingerprint,planCountQuestionPhase} from './helpers/claude-pty-runner'; -import type {NativePlanQuestionCall} from './helpers/plan-count-transcript'; -import {E2E_TOUCHFILES} from './helpers/touchfiles-data'; -const fresh=()=>structuredClone(captured.call) as NativePlanQuestionCall; -const fp=(c=fresh())=>nativePlanCallFingerprint(c,Date.parse(c.answeredAt!),true); -const first=(c=fresh())=>engFirstReviewAUQ(fp(c)); -function change(edit:(q:NativePlanQuestionCall['questions'][number])=>void){const c=fresh(),q=c.questions[0]!,picked=q.options.findIndex(o=>o.label===c.answers[q.question]);edit(q);c.answers={[q.question]:q.options[picked]!.label};return c;} -test('exact acknowledged annotated cache finding opens review without changing native ownership',()=>{ - const c=fresh(),before=JSON.stringify(c);expect(first(c)).toBe(true);expect(engSetupAUQ(fp(c))).toBe(false);expect(planCountQuestionPhase(fp(c),false,engStep0Boundary,engFirstReviewAUQ,engSetupAUQ)).toMatchObject({preReview:false,reviewStarted:true});expect(JSON.stringify(c)).toBe(before);expect(captured.provenance.retrospectivePass).toBe(false); -}); -test('incidental metadata and all offered choices retain substantive review identity',()=>{ - for(const edit of [ - (q:any)=>{q.question=q.question.replace('D5 — Issue 1','D15 — Issue 11');q.header='Arch 11';}, - (q:any)=>{q.question=q.question.replace('PLAN.md:19-20 + :10','docs/plan.md:42');}, - (q:any)=>{q.question=q.question.replace('[P1] (confidence 8/10)','[P2] (confidence 10/10)');}, - (q:any)=>{q.question=q.question.replaceAll('AuthCache','TenantStore').replaceAll('SessionMint','SessionWriter').replaceAll('AuthBroker','AuthReader');q.options=q.options.map((o:any)=>({...o,description:o.description.replaceAll('AuthCache','TenantStore').replaceAll('SessionMint','SessionWriter').replaceAll('AuthBroker','AuthReader')}));}, - (q:any)=>{q.question+='\n"Historical note: This finding is withdrawn."';}, - (q:any)=>{q.options[0].description+='\n"This option is withdrawn."';}, - ])expect(first(change(edit))).toBe(true); - for(const reversed of [false,true])for(let i=0;i<3;i++){const c=fresh(),q=c.questions[0]!;if(reversed)q.options.reverse();c.answers={[q.question]:q.options[i]!.label};expect(first(c)).toBe(true);} -}); -const changes:Array<[string,(q:NativePlanQuestionCall['questions'][number])=>void]>=[ - ['foreign issue header',q=>{q.header='Arch 2';}],['missing issue',q=>{q.question=q.question.replace('Issue 1 ','');}],['missing annotation',q=>{q.question=q.question.replace('[P1] (confidence 8/10) ','');}],['missing source location',q=>{q.question=q.question.replace('PLAN.md:19-20 + :10 — ','');}],['invalid confidence',q=>{q.question=q.question.replace('confidence 8/10','confidence 11/10');}], - ['conditional defect',q=>{q.question=q.question.replace('both mutate','might both mutate');}],['same actor twice',q=>{q.question=q.question.replace('AuthBroker and SessionMint','AuthBroker and AuthBroker');}],['serialized title',q=>{q.question=q.question.replace('does not serialize mutations','serializes mutations');}], - ['source title',q=>{q.question='Source: '+q.question;}],['quoted title',q=>{const lines=q.question.split('\n');lines[0]='"'+lines[0]+'"';q.question=lines.join('\n');}],['source context',q=>{q.question=q.question.replace('Project/branch/task:','Source:');}],['historical context',q=>{q.question=q.question.replace('Project/branch/task:','Project/branch/task: Historical assessment:');}], - ['no own explanation',q=>{q.question=q.question.replace(/^ELI10:.*$/m,'');}],['quoted explanation',q=>{q.question=q.question.replace(/^ELI10: (.*)$/m,'ELI10: "$1"');}],['competing explanation',q=>{q.question+='\nELI10: There is no race.';}],['hypothetical explanation',q=>{q.question=q.question.replace('ELI10:','ELI10: If approved,');}],['missing race consequence',q=>{q.question=q.question.replace('the mint can land after the invalidation and a suspended tenant keeps a live session','the tenant always loses the session');}], - ['repair wrong cache',q=>{q.options[0]!.description=q.options[0]!.description!.replace('AuthCache passed','OtherCache passed');}],['same writer and reader',q=>{q.options[0]!.description=q.options[0]!.description!.replace('AuthBroker reads','SessionMint reads');}],['missing invalidation rejection',q=>{q.options[0]!.description=q.options[0]!.description!.replace('are rejected if the entry was invalidated since read','are accepted even when invalidated');}],['missing owned repair',q=>{q.options[0]!.description='Choose later.';}],['missing opposed risk',q=>{q.options[2]!.description='The race is closed.';}],['opposition now serialized',q=>{q.options[2]!.description+='\nThe writers are now serialized.';}],['reader also writes',q=>{q.options[0]!.description+='\nAuthBroker also writes.';}], -]; -test.each(changes)('%s cannot open review',(_,edit)=>expect(first(change(edit))).toBe(false)); -test('current statuses, framing and conditional approval are enforced on finding and offered outcomes',()=>{ - for(const status of ['withdrawn','no longer current','hypothetical','optional'])for(const [open,close]of [['',''],['"','"'],["'","'"],['“','”'],['‘','’'],['`','`']]){ - for(const owner of ['This finding','D5','Issue 1'])expect(first(change(q=>{q.question+=`\n**${owner}** is ${open}${status}${close}.`;})),`${owner} ${open}${status}`).toBe(false); - for(const i of [0,1,2])expect(first(change(q=>{q.options[i]!.description+=`\n**This option** is ${open}${status}${close}.`;}))).toBe(false); - } - for(const prefix of ['Source:','Historical assessment:','If approved,','Once approved,','Pending approval:'])for(const i of [0,1,2])expect(first(change(q=>{q.options[i]!.description=prefix+'\n'+q.options[i]!.description;})),prefix).toBe(false); -}); -test('native completion, timestamp, exact answer, session and visible menu stay mandatory',()=>{ - const edits:Array<(c:NativePlanQuestionCall)=>void>=[c=>{c.answered=false;},c=>{c.failed=true;},c=>{c.answers={};},c=>{c.answers[c.questions[0]!.question]='not offered';},c=>{c.unansweredQuestionIndices=[0];},c=>{c.answeredAt='invalid';},c=>{c.sessionId='';},c=>{c.toolUseId='';},c=>{c.questions[0]!.multiSelect=true;},c=>{c.questions.push(structuredClone(c.questions[0]!));},c=>{c.questions[0]!.options[1]!.label=c.questions[0]!.options[0]!.label;}]; - for(const edit of edits){const c=fresh();edit(c);expect(first(c)).toBe(false);}const f=fp();expect(engFirstReviewAUQ({...f,signature:'foreign'})).toBe(false);expect(engFirstReviewAUQ({...f,options:f.options.slice().reverse()})).toBe(false);expect(engFirstReviewAUQ({...f,nativeQuestionIndex:1})).toBe(false); -}); -test('regression fixture and control select the engineering finding-count workflow',()=>{ - for(const file of ['test/eng-annotated-cache-au.test.ts','test/fixtures/eng-annotated-cache-au.json'])expect(Object.entries(E2E_TOUCHFILES).filter(([,paths])=>paths.includes(file)).map(([owner])=>owner)).toEqual([]); -}); - -test('current approval conditions and same-option effort boundaries cannot hide withdrawals',()=>{ - for(const phrase of ['requires approval','is conditional on approval','is contingent on acceptance']) for(const target of ['finding','option']) expect(first(change(q=>{if(target==='finding')q.question+='\nThis finding '+phrase+'.';else q.options[0]!.description+='\nThis option '+phrase+'.';}))).toBe(false); - for(const status of ['withdrawn','no longer current']) for(const [open,close]of [['',''],['"','"'],["'","'"],['“','”'],['‘','’']]) expect(first(change(q=>{q.options[0]!.description=q.options[0]!.description!.replace(/\.$/,'')+` This option is ${open}${status}${close}.`;}))).toBe(false); - expect(first(change(q=>{q.options[2]!.description+='\nOnly SessionMint writes.';}))).toBe(false); - expect(first(change(q=>{q.options[0]!.description+='\nDo not inject the cache.';}))).toBe(false); -}); - -test('the injection-only alternative must retain its stated unresolved race',()=>{ - for(const text of ['AuthCache is now serialized.','Only SessionMint writes.']) expect(first(change(q=>{q.options[1]!.description+='\n'+text;}))).toBe(false); - for(const text of ['"AuthCache is now serialized."',"'Only SessionMint writes.'",'ArchiveCache is now serialized.']) expect(first(change(q=>{q.options[1]!.description+='\n'+text;}))).toBe(true); -}); diff --git a/test/eng-architecture-cache-av.test.ts b/test/eng-architecture-cache-av.test.ts deleted file mode 100644 index e67f1365d..000000000 --- a/test/eng-architecture-cache-av.test.ts +++ /dev/null @@ -1,107 +0,0 @@ -import {describe, expect, test} from 'bun:test'; -import fixture from './fixtures/eng-architecture-cache-av-calls.json'; -import {engFirstReviewAUQ, engSetupAUQ, engStep0Boundary, nativePlanCallFingerprint, planCountQuestionPhase} from './helpers/claude-pty-runner'; -import type {NativePlanQuestionCall} from './helpers/plan-count-transcript'; -import {E2E_TOUCHFILES} from './helpers/touchfiles-data'; -const fresh=()=>structuredClone(fixture.call) as NativePlanQuestionCall; -const fp=(c:NativePlanQuestionCall)=>nativePlanCallFingerprint(c,0,true); -const accepted=(c:NativePlanQuestionCall)=>engFirstReviewAUQ(fp(c)); -type Q=NativePlanQuestionCall['questions'][number]; -function edit(change:(q:Q,c:NativePlanQuestionCall)=>void){const c=fresh(),q=c.questions[0]!;change(q,c);c.answers={[q.question]:q.options[0]!.label};return c;} - -describe('declarative architecture issue owns the current cache mutation decision',()=>{ - test('the exact completed public decision establishes review before the counter records it',()=>{ - const c=fresh(),before=JSON.stringify(c); - expect(c.toolUseId).toBe('toolu_0147MKgbsvnFruWMDXzQUGVv'); - expect(c.answeredAt).toBe('2026-09-10T23:03:42.025Z'); - expect(accepted(c)).toBe(true); - expect(engSetupAUQ(fp(c))).toBe(false); - expect(planCountQuestionPhase(fp(c),false,engStep0Boundary,engFirstReviewAUQ,engSetupAUQ)).toEqual({preReview:false,reviewStarted:true}); - expect(JSON.stringify(c)).toBe(before); - }); - test('actor and cache renaming, citation changes, decision ordinals and offered deferral keep meaning',()=>{ - const rename=JSON.parse(JSON.stringify(fresh()).replaceAll('AuthBroker','CredentialReader').replaceAll('SessionMint','SessionWriter').replaceAll('AuthCache','TenantCache')); - expect(accepted(rename)).toBe(true); - expect(accepted(edit(q=>{q.question=q.question.replaceAll('PLAN.md:19-20','docs/REVISED.md:31-33').replaceAll('PLAN.md:10','docs/REVISED.md:12');}))).toBe(true); - expect(accepted(edit(q=>{q.header='Arch 7';q.question=q.question.replace('D3 — Architecture issue 1','D22 — Architecture issue 7').replace(/\b1([ABC])\b/g,'7$1');q.options.forEach(o=>{o.label=o.label.replace(/^1/,'7');});}))).toBe(true); - expect(accepted(edit(q=>{q.question=q.question.replace('unserialized mutations\n','unserialized mutations.\n');}))).toBe(true); - expect(accepted(edit(q=>q.options.reverse()))).toBe(true); - for(const option of fresh().questions[0]!.options){const c=fresh();c.answers={[c.questions[0]!.question]:option.label};expect(accepted(c)).toBe(true);} - }); - test('the common native completion and identity gates remain necessary',()=>{ - for(const mutation of [ - (c:NativePlanQuestionCall)=>{c.answered=false;},(c:NativePlanQuestionCall)=>{c.failed=true;}, - (c:NativePlanQuestionCall)=>{delete c.answeredAt;},(c:NativePlanQuestionCall)=>{c.answeredAt='invalid';}, - (c:NativePlanQuestionCall)=>{c.unansweredQuestionIndices=[0];},(c:NativePlanQuestionCall)=>{c.answers={};}, - (c:NativePlanQuestionCall)=>{c.answers={[c.questions[0]!.question]:'unoffered'};}, - (c:NativePlanQuestionCall)=>{c.questions[0]!.multiSelect=true;}, - (c:NativePlanQuestionCall)=>{c.questions.push(structuredClone(c.questions[0]!));}, - ]){const c=fresh();mutation(c);expect(accepted(c)).toBe(false);} - for(const mutation of [ - (f:ReturnType)=>{f.signature='foreign:request';},(f:ReturnType)=>{f.nativeCall!.sessionId='foreign';}, - (f:ReturnType)=>{f.nativeCall!.toolUseId='foreign';},(f:ReturnType)=>{f.nativeQuestionIndex=1;}, - (f:ReturnType)=>{f.options.reverse();}, - ]){const f=fp(fresh());mutation(f);expect(engFirstReviewAUQ(f)).toBe(false);} - }); - test('finding metadata and the current shared-cache premise must agree',()=>{ - for(const mutation of [ - (q:Q)=>{q.header='Arch 2';},(q:Q)=>{q.header='Scope';}, - (q:Q)=>{q.options[0]!.label=q.options[0]!.label.replace(/^1A/,'2A');}, - (q:Q)=>{q.question=q.question.replace('global mutable AuthCache','global mutable OtherCache');}, - (q:Q)=>{q.question=q.question.replace('has AuthBroker and SessionMint','has AuthBroker and AuthBroker');}, - (q:Q)=>{q.question=q.question.replace('nothing serializes','the queue serializes');}, - (q:Q)=>{q.question=q.question.replace('ELI10: PLAN.md','ELI10: If approved, PLAN.md');}, - (q:Q)=>{q.question=q.question.replace(/^ELI10: (.+)$/m,'ELI10: "$1"');}, - (q:Q)=>{q.question=q.question.replace(/^ELI10: (.+)$/m,'> ELI10: $1');}, - (q:Q)=>{q.question='Source example:\n'+q.question;}, - (q:Q)=>{q.question='```text\n'+q.question+'\n```';}, - (q:Q)=>{q.question+='\nELI10: No current race remains.';}, - (q:Q)=>{q.question=q.question.replace('Picture SessionMint','Picture OtherWriter');}, - (q:Q)=>{q.question=q.question.replace('refreshed token for tenant A','refreshed token for tenant B');}, - ])expect(accepted(edit(mutation))).toBe(false); - }); - test('the same offered remedy must inject the named cache, serialize its writes and require tenant identity',()=>{ - for(const mutation of [ - (q:Q)=>{q.options[0]!.label=q.options[0]!.label.replace('Inject AuthCache','Inject OtherCache');}, - (q:Q)=>{q.options[0]!.label=q.options[0]!.label.replace('by constructor','through a global export');}, - (q:Q)=>{q.options[0]!.label=q.options[0]!.label.replace('owns all writes','accepts unowned writes');}, - (q:Q)=>{q.options[0]!.label=q.options[0]!.label.replace('serializes per tenant key','leaves writes unordered');}, - (q:Q)=>{q.options[0]!.label=q.options[0]!.label.replace('requires tenant context','allows missing tenant context');}, - (q:Q)=>{q.options[0]!.description=q.options[0]!.description!.replace('run in order through one owner','run concurrently through both services');}, - (q:Q)=>{q.options[0]!.description=q.options[0]!.description!.replace('fresh AuthCache per case','shared AuthCache for all cases');}, - (q:Q)=>{q.options[0]!.description=q.options[0]!.description!.replace('No method accepts a call without','Every method accepts a call without');}, - (q:Q)=>{q.options[0]!.description='Historical example: '+q.options[0]!.description;}, - (q:Q)=>{q.options[0]!.description='If approved: '+q.options[0]!.description;}, - (q:Q)=>{q.options[0]!.description='> '+q.options[0]!.description;}, - (q:Q)=>{q.options[0]!.description+='\nAuthBroker still writes directly.';}, - (q:Q)=>{q.options[0]!.description+='\nSerialization is optional.';}, - ])expect(accepted(edit(mutation))).toBe(false); - }); - test('the opposed choice must actually leave the current race open',()=>{ - for(const mutation of [ - (q:Q)=>{q.options[2]!.label='1C: Resolve the race';}, - (q:Q)=>{q.options[2]!.description='The cache is already safe and serialized.';}, - (q:Q)=>{q.options[2]!.description='Historical example: '+q.options[2]!.description;}, - (q:Q)=>{q.options[2]!.description+='\nAuthCache is already serialized.';}, - (q:Q)=>{q.options[2]!.description+='\nOnly AuthBroker writes.';}, - (q:Q)=>{q.options[1]!.description+='\nOnly AuthBroker writes.';}, - (q:Q)=>{q.question+='\nAuthCache now serializes all writes.';}, - (q:Q)=>{q.question+='\nOnly SessionMint writes.';}, - (q:Q)=>{q.question+='\nDo not inject this cache.';}, - ])expect(accepted(edit(mutation))).toBe(false); - }); - test('owned current statuses and approvals override the earlier finding across scalar quote forms',()=>{ - for(const target of [-1,0,1,2])for(const owner of ['This finding','D3','Architecture issue 1'])for(const suffix of [" is 'withdrawn'.",' is “no longer current”.',' is `unproven`.',' is optional.',' requires approval.']){ - const c=edit(q=>{const text='\nAssessment complete; '+owner+suffix;if(target<0)q.question+=text;else q.options[target]!.description+=text;}); - expect(accepted(c)).toBe(false); - } - for(const target of [-1,0,2])for(const text of ['\nPrior note: "This finding is withdrawn."','\n> This finding is withdrawn.','\nA previous reviewer said `This finding is withdrawn.`','\nOtherCache is already serialized.']){ - expect(accepted(edit(q=>{if(target<0)q.question+=text;else q.options[target]!.description+=text;}))).toBe(true); - } - }); - test('the exact public fixture and focused regression select only the Eng finding-count workflow',()=>{ - for(const dependency of ['test/eng-architecture-cache-av.test.ts','test/fixtures/eng-architecture-cache-av-calls.json']){ - expect(Object.entries(E2E_TOUCHFILES).filter(([,paths])=>paths.includes(dependency)).map(([name])=>name)).toEqual([]); - } - }); -}); diff --git a/test/eng-binding-retry-z.test.ts b/test/eng-binding-retry-z.test.ts deleted file mode 100644 index dce4baa74..000000000 --- a/test/eng-binding-retry-z.test.ts +++ /dev/null @@ -1,103 +0,0 @@ -import { describe, expect, test } from 'bun:test'; -import captured from './fixtures/eng-binding-retry-z-calls.json'; -import { engFirstReviewAUQ, engSetupAUQ, engStep0Boundary, nativePlanCallFingerprint, planCountQuestionPhase } from './helpers/claude-pty-runner'; -import type { NativePlanQuestionCall } from './helpers/plan-count-transcript'; - -const fresh = () => structuredClone(captured[1]!) as NativePlanQuestionCall; -const fp = (c: NativePlanQuestionCall) => nativePlanCallFingerprint(c, 0, true); -const first = (c: NativePlanQuestionCall) => engFirstReviewAUQ(fp(c)); -function question(c: NativePlanQuestionCall, transform: (s: string) => string) { - const q = c.questions[0]!; const answer = c.answers![q.question]!; - q.question = transform(q.question); c.answers = {[q.question]: answer}; return c; -} - -describe('Z Eng shared mutable cache starts substantive review', () => { - test('the actual shared mutable cache risk starts review without an issue label', () => { - expect(engSetupAUQ(fp(fresh()))).toBe(false); - expect(first(fresh())).toBe(true); - expect(planCountQuestionPhase(fp(fresh()), false, engStep0Boundary, engFirstReviewAUQ, engSetupAUQ)) - .toEqual({preReview: false, reviewStarted: true}); - }); - - test('the exact six native calls preserve one setup and all five review obligations', () => { - let started = false; - const phases = captured.map(c => { - const p = planCountQuestionPhase(fp(structuredClone(c) as NativePlanQuestionCall), started, engStep0Boundary, engFirstReviewAUQ, engSetupAUQ); - started = p.reviewStarted; return p.preReview; - }); - expect(phases).toEqual([true, false, false, false, false, false]); - expect(first(structuredClone(captured[0]!) as NativePlanQuestionCall)).toBe(false); - expect(captured[5]!.questions[0]!.header).toBe('TODO: E2E test'); - }); - - test('either offered choice, reordering and a different component retain issue identity', () => { - const c = fresh(); c.questions[0]!.options.reverse(); - for (const option of c.questions[0]!.options) { - c.answers = {[c.questions[0]!.question]: option.label}; expect(first(c)).toBe(true); - } - const varied = question(fresh(), s => s.replace('AuthCache', 'SessionCache').replace('D2', 'D7')); - for (const option of varied.questions[0]!.options) option.description = option.description.replaceAll('AuthCache', 'SessionCache'); - expect(first(varied)).toBe(true); - }); - - test('requires a complete native call and exact offered answer and fingerprint', () => { - for (const mutate of [ - (c: NativePlanQuestionCall) => { c.answered = false; }, - (c: NativePlanQuestionCall) => { c.failed = true; }, - (c: NativePlanQuestionCall) => { delete c.failed; }, - (c: NativePlanQuestionCall) => { delete c.unansweredQuestionIndices; }, - (c: NativePlanQuestionCall) => { c.unansweredQuestionIndices = [0]; }, - (c: NativePlanQuestionCall) => { c.questions[0]!.multiSelect = true; }, - (c: NativePlanQuestionCall) => { c.questions.push(structuredClone(c.questions[0]!)); }, - (c: NativePlanQuestionCall) => { c.questions[0]!.options.push(structuredClone(c.questions[0]!.options[0]!)); }, - (c: NativePlanQuestionCall) => { c.questions[0]!.options[1]!.label = c.questions[0]!.options[0]!.label; }, - (c: NativePlanQuestionCall) => { c.answers = {}; }, - (c: NativePlanQuestionCall) => { c.answers = {[c.questions[0]!.question]: 'Foreign answer'}; }, - ]) { const c = fresh(); mutate(c); expect(first(c)).toBe(false); } - expect(engFirstReviewAUQ({...fp(fresh()), signature: 'foreign:call'})).toBe(false); - expect(engFirstReviewAUQ({...fp(fresh()), nativeCall: undefined})).toBe(false); - expect(engFirstReviewAUQ({...fp(fresh()), options: []})).toBe(false); - const mismatch = fp(fresh()); mismatch.options[0]!.label = 'foreign'; expect(engFirstReviewAUQ(mismatch)).toBe(false); - const wrongIndex = fp(fresh()); wrongIndex.options[0]!.index = 2; expect(engFirstReviewAUQ(wrongIndex)).toBe(false); - }); - - test('requires an affirmative direct risk, not setup, denial, qualification or quotations', () => { - for (const header of ['Scope', 'Approach', 'Next review', 'Onboarding']) { - const c = fresh(); c.questions[0]!.header = header; expect(first(c)).toBe(false); - } - for (const transform of [ - (s: string) => s.replace('Architecture:', 'Approach:'), - (s: string) => s.replace('Two services share', 'If two services share'), - (s: string) => s.replace('Two services share', 'Two services do not share'), - (s: string) => s.replace('can corrupt tenant isolation', 'cannot corrupt tenant isolation'), - (s: string) => s.replace('can corrupt tenant isolation', 'never corrupt tenant isolation'), - (s: string) => s.replace('This is the #1 reliability risk', 'This is not the #1 reliability risk'), - (s: string) => s.replace('plan-eng-shared-mutable-cache', 'plan-eng-setup'), - (s: string) => s.replace('plan-eng-shared-mutable-cache', 'foreign-shared-mutable-cache'), - (s: string) => s.replace(/ ]+>/, ''), - (s: string) => s + ' ', - (s: string) => s + ' Run the next review too.', - (s: string) => '> ' + s, - (s: string) => '```text\n' + s + '\n```', - ]) expect(first(question(fresh(), transform))).toBe(false); - }); - - test('the complete offered remedies stay tied to the same dependency and affirmative risk', () => { - for (const [index, transform] of [ - [0, (s: string) => s.replace('The plan is updated', 'The plan is not updated')], - [0, (s: string) => s.replace('pass AuthCache', 'pass DifferentCache')], - [0, (s: string) => s.replace('No module-level mutable export.', 'Keep the module-level mutable export.')], - [1, (s: string) => s.replace('still couples both services', 'does not couple both services')], - [2, (s: string) => s.replace('as a known risk', 'as a dismissed risk')], - [0, (s: string) => s + ' Also grant every tenant access.'], - [1, (s: string) => s + ' Also approve the missing timeout policy.'], - [0, (s: string) => '> ' + s], - [2, (s: string) => '```text\n' + s + '\n```'], - ] as const) { - const c = fresh(); const option = c.questions[0]!.options[index]!; - option.description = transform(option.description ?? ''); expect(first(c)).toBe(false); - } - const c = fresh(); c.questions[0]!.options[0]!.label = 'Run /office-hours'; - c.answers = {[c.questions[0]!.question]: 'Run /office-hours'}; expect(first(c)).toBe(false); - }); -}); diff --git a/test/eng-binding-z.test.ts b/test/eng-binding-z.test.ts deleted file mode 100644 index f6b73f67d..000000000 --- a/test/eng-binding-z.test.ts +++ /dev/null @@ -1,101 +0,0 @@ -import { describe, expect, test } from 'bun:test'; -import captured from './fixtures/eng-binding-z-calls.json'; -import { engFirstReviewAUQ, engSetupAUQ, engStep0Boundary, nativePlanCallFingerprint, planCountQuestionPhase } from './helpers/claude-pty-runner'; -import type { NativePlanQuestionCall } from './helpers/plan-count-transcript'; - -const fresh = () => structuredClone(captured[1]!) as NativePlanQuestionCall; -const fp = (c: NativePlanQuestionCall) => nativePlanCallFingerprint(c, 0, true); -const first = (c: NativePlanQuestionCall) => engFirstReviewAUQ(fp(c)); -function question(c: NativePlanQuestionCall, transform: (s: string) => string) { - const q = c.questions[0]!; const answer = c.answers![q.question]!; - q.question = transform(q.question); c.answers = {[q.question]: answer}; return c; -} - -describe('Z Eng dependency binding starts substantive review', () => { - test('the actual cache dependency decision starts review without an issue label', () => { - expect(engSetupAUQ(fp(fresh()))).toBe(false); - expect(first(fresh())).toBe(true); - expect(planCountQuestionPhase(fp(fresh()), false, engStep0Boundary, engFirstReviewAUQ, engSetupAUQ)) - .toEqual({preReview: false, reviewStarted: true}); - }); - - test('the exact six native calls preserve one setup and all five review obligations', () => { - let started = false; - const phases = captured.map(c => { - const p = planCountQuestionPhase(fp(structuredClone(c) as NativePlanQuestionCall), started, engStep0Boundary, engFirstReviewAUQ, engSetupAUQ); - started = p.reviewStarted; return p.preReview; - }); - expect(phases).toEqual([true, false, false, false, false, false]); - expect(first(structuredClone(captured[0]!) as NativePlanQuestionCall)).toBe(false); - expect(captured[5]!.questions[0]!.header).toBe('TODO: Timeout'); - }); - - test('either offered choice, reordering and a different component retain issue identity', () => { - const c = fresh(); c.questions[0]!.options.reverse(); - for (const option of c.questions[0]!.options) { - c.answers = {[c.questions[0]!.question]: option.label}; expect(first(c)).toBe(true); - } - const varied = question(fresh(), s => s.replace('AuthBroker', 'SessionGateway').replace('D2', 'D7')); - for (const option of varied.questions[0]!.options) option.description = option.description.replaceAll('AuthBroker', 'SessionGateway'); - expect(first(varied)).toBe(true); - }); - - test('requires a complete native call and exact offered answer and fingerprint', () => { - for (const mutate of [ - (c: NativePlanQuestionCall) => { c.answered = false; }, - (c: NativePlanQuestionCall) => { c.failed = true; }, - (c: NativePlanQuestionCall) => { delete c.failed; }, - (c: NativePlanQuestionCall) => { delete c.unansweredQuestionIndices; }, - (c: NativePlanQuestionCall) => { c.unansweredQuestionIndices = [0]; }, - (c: NativePlanQuestionCall) => { c.questions[0]!.multiSelect = true; }, - (c: NativePlanQuestionCall) => { c.questions.push(structuredClone(c.questions[0]!)); }, - (c: NativePlanQuestionCall) => { c.questions[0]!.options.push(structuredClone(c.questions[0]!.options[0]!)); }, - (c: NativePlanQuestionCall) => { c.questions[0]!.options[1]!.label = c.questions[0]!.options[0]!.label; }, - (c: NativePlanQuestionCall) => { c.answers = {}; }, - (c: NativePlanQuestionCall) => { c.answers = {[c.questions[0]!.question]: 'Foreign answer'}; }, - ]) { const c = fresh(); mutate(c); expect(first(c)).toBe(false); } - expect(engFirstReviewAUQ({...fp(fresh()), signature: 'foreign:call'})).toBe(false); - expect(engFirstReviewAUQ({...fp(fresh()), nativeCall: undefined})).toBe(false); - expect(engFirstReviewAUQ({...fp(fresh()), options: []})).toBe(false); - const mismatch = fp(fresh()); mismatch.options[0]!.label = 'foreign'; expect(engFirstReviewAUQ(mismatch)).toBe(false); - const wrongIndex = fp(fresh()); wrongIndex.options[0]!.index = 2; expect(engFirstReviewAUQ(wrongIndex)).toBe(false); - }); - - test('setup, foreign, hypothetical, quoted and mixed questions cannot open review', () => { - for (const header of ['Scope', 'Approach', 'Next review', 'Onboarding']) { - const c = fresh(); c.questions[0]!.header = header; expect(first(c)).toBe(false); - } - for (const transform of [ - (s: string) => s.replace('Architecture:', 'Approach:'), - (s: string) => s.replace('How should AuthBroker', 'If needed, how should AuthBroker'), - (s: string) => s.replace('AuthBroker access', 'the whole plan access'), - (s: string) => s.replace('plan-eng-cache-binding', 'plan-eng-setup'), - (s: string) => s.replace('plan-eng-cache-binding', 'foreign-cache-binding'), - (s: string) => s.replace(/ ]+>/, ''), - (s: string) => s + ' ', - (s: string) => s + ' Approve the release too.', - (s: string) => '> ' + s, - (s: string) => '```text\n' + s + '\n```', - ]) expect(first(question(fresh(), transform))).toBe(false); - }); - - test('both descriptions must affirm the existing dependency and remedy without extra obligations', () => { - for (const [index, transform] of [ - [0, (s: string) => s.replace('Eliminates module-level mutable state entirely.', 'Does not eliminate module-level mutable state.')], - [0, (s: string) => s.replace('Eliminates module-level mutable state entirely.', 'If shared state exists, eliminates it.')], - [0, (s: string) => s.replace('AuthBroker receives', 'DifferentComponent receives')], - [1, (s: string) => s.replace('same pattern as the current plan', 'unlike the current plan')], - [1, (s: string) => s.replace('makes tests require module-level mocking', 'does not make tests require module-level mocking')], - [1, (s: string) => s.replace('AuthBroker imports', 'DifferentComponent imports')], - [0, (s: string) => s + ' Also grant every tenant access.'], - [1, (s: string) => s + ' Also approve the missing timeout policy.'], - [0, (s: string) => '> ' + s], - [1, (s: string) => '```text\n' + s + '\n```'], - ] as const) { - const c = fresh(); const option = c.questions[0]!.options[index]!; - option.description = transform(option.description ?? ''); expect(first(c)).toBe(false); - } - const c = fresh(); c.questions[0]!.options[0]!.label = 'Run /office-hours'; - c.answers = {[c.questions[0]!.question]: 'Run /office-hours'}; expect(first(c)).toBe(false); - }); -}); diff --git a/test/eng-cache-brief-am.test.ts b/test/eng-cache-brief-am.test.ts deleted file mode 100644 index 4b8df60e1..000000000 --- a/test/eng-cache-brief-am.test.ts +++ /dev/null @@ -1,56 +0,0 @@ -import {expect,test} from 'bun:test'; -import {engFirstReviewAUQ,nativePlanCallFingerprint,type AskUserQuestionFingerprint as FP} from './helpers/claude-pty-runner'; -import fixture from './fixtures/eng-cache-brief-am.json'; -const calls=fixture.calls as FP[]; -function edit(change:(q:any,f:FP)=>void):FP { const f=structuredClone(calls[1]!),c=f.nativeCall!,q=c.questions[0]!,selected=q.options.findIndex(o=>o.label===c.answers?.[q.question]);change(q,f);c.answers={[q.question]:q.options[selected]!.label};return nativePlanCallFingerprint(c,f.observedAtMs,f.preReview); } -test('the completed current cache ownership brief starts the engineering review',()=>expect(engFirstReviewAUQ(calls[1]!)).toBe(true)); -test('the earlier whole-plan scope choice does not become a finding',()=>expect(engFirstReviewAUQ(calls[0]!)).toBe(false)); -test('equivalent decision ordinal and current wording retain the owned finding',()=>{ - expect(engFirstReviewAUQ(edit(q=>{q.question=q.question.replace(/^D2/,'D17');q.header='D17 DI';}))).toBe(true); - expect(engFirstReviewAUQ(edit(q=>{q.question=q.question.replace('Right now both services grab','Today both services import').replace('and both write to it.','and both mutate it.');}))).toBe(true); -}); -test('unrelated current decisions remain outside this dependency branch',()=>{ - expect(engFirstReviewAUQ(edit(q=>{q.header='D3 DI';}))).toBe(false); - expect(engFirstReviewAUQ(edit(q=>{q.question=q.question.replace('Architecture finding A1','Architecture finding A2');}))).toBe(false); -}); -const negative:Array<[string,(q:any,f:FP)=>void]>=[ - ['source preface',q=>q.question=q.question.replace('\nELI10:','\nSource excerpt:\nELI10:')], - ['historical current clause',q=>q.question=q.question.replace('ELI10: Right now','ELI10: Previously')], - ['hypothetical current clause',q=>q.question=q.question.replace('ELI10: Right now','ELI10: If approved, right now')], - ['negated current writes',q=>q.question=q.question.replace('and both write to it.','and neither writes to it.')], - ['quoted current assessment',q=>q.question=q.question.replace('ELI10: Right now','ELI10: "Right now').replace('it. Nobody','it." Nobody')], - ['withdrawn finding',q=>q.question+='\nCorrection: this finding is withdrawn.'], - ['resolved current finding',q=>q.question+='\nNo current gap remains.'], - ['foreign cache title',q=>q.question=q.question.replace('Module-level AuthCache','Module-level OtherCache')], - ['source remedy preface',q=>q.options[0].description='Source excerpt:\n'+q.options[0].description], - ['conditional writer ownership',q=>q.options[0].description=q.options[0].description.replace('✅ SessionMint','✅ If SessionMint')], - ['quoted writer ownership',q=>q.options[0].description=q.options[0].description.replace('✅ SessionMint','✅ "SessionMint').replace('not convention.','not convention."')], - ['read-only claim only in con',q=>q.options[0].description=q.options[0].description.replace('✅ SessionMint','❌ SessionMint')], - ['same writable and read-only actor',q=>q.options[0].description=q.options[0].description.replace('AuthBroker gets','SessionMint gets')], - ['withdrawn remedy',q=>q.options[0].description+=' This remedy is withdrawn.'], - ['foreign opposing finding',q=>q.options[2].description=q.options[2].description.replace('Both A1','Both A9')], - ['opposed gap resolved',q=>q.options[2].description+=' No current gap remains.'], - ['conditional opposed gap',q=>q.options[2].description=q.options[2].description.replace('❌ Both A1','❌ If Both A1')], - ['unoffered recommendation',q=>q.question=q.question.replace('Recommendation: A','Recommendation: D')], - ['multiple recommendations',q=>q.options[1].label+=' (recommended)'], - ['unlettered choice',q=>q.options[1].label=q.options[1].label.slice(3)], - ['unanswered',(_,f)=>f.nativeCall!.answered=false], - ['failed',(_,f)=>f.nativeCall!.failed=true], - ['incomplete member',(_,f)=>f.nativeCall!.unansweredQuestionIndices=[0]], - ['invalid completion time',(_,f)=>f.nativeCall!.answeredAt='invalid'], -]; -test.each(negative)('%s cannot supply current owned engineering review',(_,change)=>expect(engFirstReviewAUQ(edit(change))).toBe(false)); -test('whole quoted history and consistent identifiers preserve the current decision',()=>{ - expect(engFirstReviewAUQ(edit(q=>q.question+='\nPrior note: "This finding is withdrawn."'))).toBe(true); - expect(engFirstReviewAUQ(edit(q=>{q.question=q.question.replaceAll('AuthCache','SessionCache').replaceAll('A1','A9');q.options.forEach((o:any)=>o.description=o.description.replaceAll('A1','A9'));}))).toBe(true); - expect(engFirstReviewAUQ(edit(q=>{q.question=q.question.replace(/^D2/,'d2');q.header='d2 DI';}))).toBe(true); -}); - -import { E2E_TOUCHFILES } from './helpers/touchfiles-data'; -test('new inputs have only the engineering finding owner and dense paths',()=>{ - for(const file of ['test/eng-cache-brief-am.test.ts','test/fixtures/eng-cache-brief-am.json']) expect(Object.entries(E2E_TOUCHFILES).filter(([,paths])=>paths.includes(file)).map(([owner])=>owner)).toEqual([]); -}); -test('current-owner withdrawals and conditional metadata cannot lend review evidence',()=>{ - for(const text of ['Correction: this finding is rejected.','Correction: this remedy is cancelled.','Correction: this finding is "withdrawn".','Correction: this explanation is not current.']) expect(engFirstReviewAUQ(edit(q=>q.question+='\n'+text))).toBe(false); - expect(engFirstReviewAUQ(edit(q=>q.question=q.question.replace('Architecture finding A1','If approved, Architecture finding A1')))).toBe(false); -}); diff --git a/test/eng-cache-owner-an.test.ts b/test/eng-cache-owner-an.test.ts deleted file mode 100644 index c8740c506..000000000 --- a/test/eng-cache-owner-an.test.ts +++ /dev/null @@ -1,91 +0,0 @@ -import { expect, test } from 'bun:test'; -import fixture from './fixtures/eng-cache-owner-an.json'; -import { engFirstReviewAUQ, engStep0Boundary, engSetupAUQ, planCountQuestionPhase } from './helpers/claude-pty-runner'; -import type { AskUserQuestionFingerprint as Fingerprint } from './helpers/claude-pty-runner'; -import { E2E_TOUCHFILES, selectTests } from './helpers/touchfiles'; - -type Question = NonNullable['questions'][number]; -const original = () => structuredClone(fixture.fingerprint) as Fingerprint; -function edit(change: (question: Question) => void): Fingerprint { - const fp = original(), call = fp.nativeCall!, question = call.questions[0]!; - const selected = question.options.findIndex(option => option.label === call.answers![question.question]); - change(question); - call.answers = { [question.question]: question.options[selected]!.label }; - fp.options = question.options.map((option, index) => ({ index: index + 1, label: option.label })); - return fp; -} - -test('an owned cache-ownership decision starts review with actors named in the current assessment', () => { - expect(engFirstReviewAUQ(original())).toBe(true); - expect(planCountQuestionPhase(original(), false, engStep0Boundary, engFirstReviewAUQ, engSetupAUQ)) - .toMatchObject({ preReview: false, reviewStarted: true }); - expect(fixture.fingerprint.preReview).toBe(true); -}); - -test('review identity survives equivalent headers, actor names and an offered opposing answer', () => { - for (const header of ['Cache owner', 'Cache ownership', 'Shared cache', 'Issue 1', 'Architecture 1']) - expect(engFirstReviewAUQ(edit(question => { question.header = header; }))).toBe(true); - expect(engFirstReviewAUQ(edit(question => { - question.question = question.question.replaceAll('AuthBroker', 'SessionOwner').replaceAll('SessionMint', 'TokenMinter'); - question.options = question.options.map(option => ({ ...option, - description: option.description?.replaceAll('AuthBroker', 'SessionOwner').replaceAll('SessionMint', 'TokenMinter') })); - }))).toBe(true); - for (const option of original().nativeCall!.questions[0]!.options) { - const fp = original(), call = fp.nativeCall!; - call.answers = { [call.questions[0]!.question]: option.label }; - expect(engFirstReviewAUQ(fp)).toBe(true); - } - expect(engFirstReviewAUQ(edit(question => { question.question += '\nHistorical note: "This finding is withdrawn."'; }))).toBe(true); -}); - -const rejected: Array<[string, (question: Question) => void]> = [ - ['setup header', q => { q.header = 'Outside voices'; }], - ['foreign issue identity', q => { q.header = 'Issue 2'; }], - ['historical title', q => { q.question = 'Historical example:\n' + q.question; }], - ['source assessment', q => { q.question = q.question.replace('\nELI10:', '\nSource:\nELI10:'); }], - ['conditional project', q => { q.question = q.question.replace('Project/branch/task: ', 'Project/branch/task: If approved: '); }], - ['quoted current premise', q => { q.question = q.question.replace(/ELI10: ([^\n]+)/, 'ELI10: "$1"'); }], - ['conditional current premise', q => { q.question = q.question.replace('ELI10: AuthBroker', 'ELI10: If AuthBroker'); }], - ['only one actual actor', q => { q.question = q.question.replace('AuthBroker and SessionMint', 'AuthBroker and AuthBroker'); }], - ['current writes negated', q => { q.question = q.question.replace('both write into', 'neither writes into'); }], - ['withdrawn finding', q => { q.question += '\nThis finding is withdrawn.'; }], - ['rejected numbered issue', q => { q.question += '\nIssue 1 is rejected.'; }], - ['assessment no longer current', q => { q.question += '\nThis assessment is not current.'; }], - ['foreign writer', q => { q.options[0]!.description = q.options[0]!.description!.replace('Only AuthBroker writes', 'Only OtherService writes'); }], - ['foreign producer', q => { q.options[0]!.description = q.options[0]!.description!.replace('SessionMint returns', 'OtherService returns'); }], - ['same writer and producer', q => { q.options[0]!.description = q.options[0]!.description!.replace('SessionMint returns', 'AuthBroker returns'); }], - ['quoted remedy', q => { q.options[0]!.description = '> ' + q.options[0]!.description; }], - ['conditional remedy', q => { q.options[0]!.description = 'If approved: ' + q.options[0]!.description; }], - ['cancelled remedy', q => { q.options[0]!.description += ' This remedy is cancelled.'; }], - ['explicitly rejected injection', q => { q.options[0]!.description += ' Correction: do not inject the adapter.'; }], - ['no opposed action', q => { q.options[2]!.label = 'C) Run another review'; }], - ['quoted deferral', q => { q.options[2]!.description = '> ' + q.options[2]!.description; }], - ['conditional deferral', q => { q.options[2]!.description = 'If approved: ' + q.options[2]!.description; }], - ['no retained race', q => { q.options[2]!.description = q.options[2]!.description!.replace('Race stays open', 'Race is closed'); }], - ['rejected opposed action', q => { q.options[2]!.description += ' This option is rejected.'; }], - ['withdrawn single-writer requirement', q => { q.options[0]!.description += ' The single-writer requirement is withdrawn.'; }], - ['producer also writes', q => { q.options[0]!.description += ' Correction: SessionMint will also write directly to the cache.'; }], - ['retained race closed', q => { q.options[2]!.description += ' Correction: the race is now closed.'; }], -]; -test.each(rejected)('%s does not establish the first review decision', (_, change) => { - expect(engFirstReviewAUQ(edit(change))).toBe(false); -}); - -test('native ownership, completion, answer alignment and dense menus remain required', () => { - const invalid: Array<(fp: Fingerprint) => void> = [ - fp => { fp.nativeCall!.answered = false; }, fp => { fp.nativeCall!.failed = true; }, - fp => { fp.signature = 'foreign:call'; }, fp => { fp.nativeQuestionIndex = 1; }, - fp => { fp.nativeCall!.unansweredQuestionIndices = [0]; }, - fp => { delete fp.nativeCall!.answeredAt; }, fp => { fp.nativeCall!.answers = {}; }, - fp => { fp.options.reverse(); }, - ]; - for (const change of invalid) { - const fp = original(); change(fp); - expect(engFirstReviewAUQ(fp)).toBe(false); - } -}); - -test('new source dependencies select only the affected engineering workflow', () => { - for (const path of ['test/eng-cache-owner-an.test.ts', 'test/fixtures/eng-cache-owner-an.json']) - expect(selectTests([path], E2E_TOUCHFILES, []).selected).toEqual([]); -}); diff --git a/test/eng-cache-writes-as.test.ts b/test/eng-cache-writes-as.test.ts deleted file mode 100644 index 47080af1b..000000000 --- a/test/eng-cache-writes-as.test.ts +++ /dev/null @@ -1,115 +0,0 @@ -import { expect, test } from 'bun:test'; -import captured from './fixtures/eng-cache-writes-as.json'; -import { engFirstReviewAUQ, engSetupAUQ, engStep0Boundary, nativePlanCallFingerprint, planCountQuestionPhase } from './helpers/claude-pty-runner'; -import type { NativePlanQuestionCall } from './helpers/plan-count-transcript'; -import { E2E_TOUCHFILES, selectTests } from './helpers/touchfiles'; - -const actual = () => structuredClone(captured.call) as NativePlanQuestionCall; -function answered(c: NativePlanQuestionCall, index = 0) { - c.answers = { [c.questions[0]!.question]: c.questions[0]!.options[index]!.label }; - return nativePlanCallFingerprint(c, 0, true); -} -function allText(edit: (s: string) => string) { - const c = actual(), q = c.questions[0]!; q.question = edit(q.question); - for (const o of q.options) { o.label = edit(o.label); o.description = edit(o.description ?? ''); } - return c; -} - -test('the exact completed retry starts review with the current cache ownership decision', () => { - const c = actual(), before = JSON.stringify(c), fp = nativePlanCallFingerprint(c, 0, true); - expect(engFirstReviewAUQ(fp)).toBe(true); expect(engSetupAUQ(fp)).toBe(false); - expect(planCountQuestionPhase(fp, false, engStep0Boundary, engFirstReviewAUQ, engSetupAUQ)).toMatchObject({ preReview: false, reviewStarted: true }); - expect(JSON.stringify(c)).toBe(before); expect(captured.provenance.retrospectivePass).toBe(false); -}); - -test('all offered choices and consistently renamed actors retain the same review identity', () => { - for (const reverse of [false, true]) for (let i = 0; i < 3; i++) { - const c = actual(); if (reverse) c.questions[0]!.options.reverse(); - expect(engFirstReviewAUQ(answered(c, i))).toBe(true); - } - for (const [one, two] of [['One', 'Two'], ['$Reader', '_Writer'], ['SessionMint', 'AuthBroker']]) { - const c = allText(t => t.replaceAll('AuthBroker', '__one__').replaceAll('SessionMint', two).replaceAll('__one__', one)); - expect(engFirstReviewAUQ(answered(c))).toBe(true); - } - const c = allText(t => t.replace(/^D2 /, 'D17 ').replace(/\b2([A-C])\b/g, '17$1')); - expect(engFirstReviewAUQ(answered(c))).toBe(true); -}); - -test('native completion, session, exact answer and option binding remain mandatory', () => { - const mutations: Array<(c: NativePlanQuestionCall) => void> = [ - c => { c.answered = false; }, c => { c.failed = true; }, c => { c.answers = {}; }, - c => { c.answers = { [c.questions[0]!.question]: 'unoffered' }; }, c => { c.answers!['foreign'] = 'answer'; }, - c => { c.answeredAt = 'invalid'; }, c => { c.unansweredQuestionIndices = [0]; }, - c => { c.sessionId = ''; }, c => { c.toolUseId = ''; }, c => { c.questions[0]!.multiSelect = true; }, - c => { c.questions.push(structuredClone(c.questions[0]!)); }, c => { c.questions[0]!.options[1]!.label = c.questions[0]!.options[0]!.label; }, - ]; - for (const mutate of mutations) { const c = actual(); mutate(c); expect(engFirstReviewAUQ(nativePlanCallFingerprint(c, 0, true))).toBe(false); } - const fp = answered(actual()); - expect(engFirstReviewAUQ({ ...fp, signature: 'foreign' })).toBe(false); - expect(engFirstReviewAUQ({ ...fp, nativeQuestionIndex: 1 })).toBe(false); - expect(engFirstReviewAUQ({ ...fp, options: [...fp.options].reverse() })).toBe(false); -}); - -test('title, own context, current assessment and two distinct writers are required', () => { - for (const edit of [ - (t: string) => 'Source.\n' + t, (t: string) => '> ' + t, (t: string) => '```\n' + t + '\n```', - (t: string) => t.replace('Who is allowed', 'Who was allowed'), - (t: string) => t.replace('the auth cache?', 'the billing cache?'), - (t: string) => t.replace('Project/branch/task:', 'Earlier review:'), - (t: string) => t.replace('AuthBroker and SessionMint both', 'AuthBroker and AuthBroker both'), - (t: string) => t.replace('both mutating one backing cache', 'both previously mutating one backing cache'), - (t: string) => t.replace('ELI10: Two services', 'ELI10: Source. Two services'), - (t: string) => t.replace('ELI10: Two services', 'ELI10: If approved, two services'), - (t: string) => t.replace('nothing orders their writes.', 'their writes are serialized.'), - (t: string) => t.replace('Project/branch/task: ', 'Project/branch/task: Assuming approval, '), - (t: string) => t.replace('Project/branch/task: ', 'Project/branch/task: Source. '), - (t: string) => t.replace('Multi-tenant Auth Refactor,', 'Multi-tenant Auth Refactor if approved,'), - ]) { const c = actual(); c.questions[0]!.question = edit(c.questions[0]!.question); expect(engFirstReviewAUQ(answered(c))).toBe(false); } - for (const header of ['Scope', 'Issue 1', 'Report', 'Cache examples']) { const c = actual(); c.questions[0]!.header = header; expect(engFirstReviewAUQ(answered(c))).toBe(false); } -}); - -test('owned current status beats a matching assertion while archived and foreign status does not', () => { - for (const status of ['withdrawn', 'superseded', 'rejected', 'cancelled', 'closed', 'hypothetical', 'not current', 'no longer current']) { - for (const [open, close] of [['', ''], ['"', '"'], ["'", "'"], ['“', '”'], ['‘', '’'], ['`', '`']]) { - for (const target of [-1, 0, 2]) for (const owner of ['This finding', 'D2']) { - const c = actual(), q = c.questions[0]!, suffix = `\nCorrection: ${owner} is ${open}${status}${close}.`; - if (target < 0) q.question += suffix; else q.options[target]!.description += suffix; - expect(engFirstReviewAUQ(answered(c)), `${target}: ${owner} ${open}${status}${close}`).toBe(false); - } - } - } - for (const tail of ['D29 is withdrawn.', '> This finding is withdrawn.', 'The prior report said "This finding is withdrawn."', 'An archived review recorded this finding is "withdrawn".', 'An archived review recorded this finding is \'withdrawn\'.', '```\nThis finding is withdrawn.\n```']) { - for (const target of [-1, 0, 2]) { const c = actual(), q = c.questions[0]!; - if (target < 0) q.question += '\n' + tail; else q.options[target]!.description += '\n' + tail; - expect(engFirstReviewAUQ(answered(c)), `${target}: ${tail}`).toBe(true); - } - } -}); - -test('a remedy and opposed choice must bind the same current writers and active race', () => { - for (const index of [0, 2]) for (const prefix of ['Source. ', 'If approved, ', 'Assuming approval, ', 'Historical assessment: ', '> ', '"']) { - const c = actual(), o = c.questions[0]!.options[index]!; o.description = prefix + o.description + (prefix === '"' ? '"' : ''); - expect(engFirstReviewAUQ(answered(c))).toBe(false); - } - for (const index of [0, 1]) for (const name of ['Foreign', 'AuthBroker']) { - const c = actual(), o = c.questions[0]!.options[index]!; o.description = o.description!.replace('SessionMint', name); - expect(engFirstReviewAUQ(answered(c))).toBe(false); - } - for (const [target, tail] of [ - [-1, 'The services no longer mutate the cache.'], [-1, 'Correction: AuthBroker no longer writes to the cache.'], - [-1, 'The writes are now serialized.'], [-1, 'The writers are now serialized.'], [2, 'Correction: The writers are now serialized.'], [0, 'Correction: AuthBroker also writes to the cache.'], - [0, 'The adapter accepts stale writes.'], [0, 'The version check is optional.'], - [2, 'The race is resolved.'], [2, 'Correction: Do not keep both writers.'], - [2, 'Only SessionMint writes to the cache.'], [2, 'Both writers no longer mutate the cache.'], - ] as const) { - const c = actual(), q = c.questions[0]!; if (target < 0) q.question += '\n' + tail; else q.options[target]!.description += '\n' + tail; - expect(engFirstReviewAUQ(answered(c)), `${target}: ${tail}`).toBe(false); - } - for (const i of [0, 2]) { const c = actual(); c.questions[0]!.options[i]!.label = `2${i ? 'C' : 'A'} Record the report`; expect(engFirstReviewAUQ(answered(c))).toBe(false); } -}); - -test('new source and exact public fixture select both Eng boundary owners', () => { - for (const file of ['test/helpers/eng-cache-writer-decision.ts', 'test/eng-cache-writes-as.test.ts', 'test/fixtures/eng-cache-writes-as.json']) { - expect(selectTests([file], E2E_TOUCHFILES, []).selected.sort()).toEqual(['plan-eng-multi-finding-batching']); - } -}); diff --git a/test/eng-count-ad-v2.test.ts b/test/eng-count-ad-v2.test.ts deleted file mode 100644 index 6197b9b48..000000000 --- a/test/eng-count-ad-v2.test.ts +++ /dev/null @@ -1,168 +0,0 @@ -import { describe, expect, test } from 'bun:test'; -import captured from './fixtures/eng-count-ad-v2.json'; -import { engFirstReviewAUQ, engSetupAUQ, engStep0Boundary, nativePlanCallFingerprint, planCountQuestionPhase } from './helpers/claude-pty-runner'; -import type { NativePlanQuestionCall } from './helpers/plan-count-transcript'; -import { E2E_TOUCHFILES, matchGlob } from './helpers/touchfiles'; -const firstCalls = captured.cases.first.calls as NativePlanQuestionCall[]; -const retryCalls = captured.cases.retry.calls as NativePlanQuestionCall[]; -const issue = () => structuredClone(retryCalls[3]!); -const fp = (call: NativePlanQuestionCall) => nativePlanCallFingerprint(call, 0, true); -const isFirst = (call: NativePlanQuestionCall) => engFirstReviewAUQ(fp(call)); -function setupPacket(): NativePlanQuestionCall { - const c = issue(); - c.questions = [ - { header: 'Design doc', question: 'No design doc found for this branch. /office-hours produces sharper review input. Run it first?', - multiSelect: false, options: [{ label: 'Skip — proceed with standard review (recommended)' }, { label: 'Run /office-hours now' }] }, - { header: 'Learnings', question: 'Search learnings from your other projects on this machine?', - multiSelect: false, options: [{ label: 'Enable cross-project learnings (recommended)' }, { label: 'Keep learnings project-scoped only' }] }, - ]; - c.answers = Object.fromEntries(c.questions.map(q => [q.question, q.options[0]!.label])); - return c; -} -function changeQuestion(call: NativePlanQuestionCall, change: (s: string) => string) { - const q = call.questions[0]!, answer = call.answers?.[q.question]; - q.question = change(q.question); call.answers = answer ? { [q.question]: answer } : {}; return call; -} -function census(calls: NativePlanQuestionCall[]) { - let reviewStarted = false; - const counts = { setup: 0, review: 0, administrative: 0 }; - const phases = calls.map(call => { - const phase = planCountQuestionPhase(fp(call), reviewStarted, engStep0Boundary, engFirstReviewAUQ, engSetupAUQ); - reviewStarted = phase.reviewStarted; - counts[phase.administrative ? 'administrative' : phase.preReview ? 'setup' : 'review']++; - return phase; - }); - return { counts, phases }; -} - -describe('Eng AD v2 completed native count evidence', () => { - test('a completed prerequisite and learnings packet closes setup without counting it as a finding', () => { - for (const reverse of [false, true]) { - const c = setupPacket(); if (reverse) c.questions.reverse(); - for (const answer of c.questions.find(q => q.header === 'Learnings')!.options) { - const learning = c.questions.find(q => q.header === 'Learnings')!; - c.answers![learning.question] = answer.label; - const phase = planCountQuestionPhase(fp(c), false, engStep0Boundary, engFirstReviewAUQ, engSetupAUQ); - expect(phase).toEqual({ preReview: true, reviewStarted: true }); - expect(planCountQuestionPhase(fp(issue()), phase.reviewStarted, - engStep0Boundary, engFirstReviewAUQ, engSetupAUQ).preReview).toBe(false); - } - } - }); - - test('partial, ambiguous, foreign and prerequisite-running packets cannot close setup', () => { - for (const mutate of [ - (c: NativePlanQuestionCall) => { c.answered = false; }, - (c: NativePlanQuestionCall) => { c.failed = true; }, - (c: NativePlanQuestionCall) => { c.unansweredQuestionIndices = [1]; }, - (c: NativePlanQuestionCall) => { delete c.answers![c.questions[1]!.question]; }, - (c: NativePlanQuestionCall) => { c.answers![c.questions[1]!.question] = 'unoffered'; }, - (c: NativePlanQuestionCall) => { c.answers![c.questions[0]!.question] = c.questions[0]!.options[1]!.label; }, - (c: NativePlanQuestionCall) => { c.questions[0]!.multiSelect = true; }, - (c: NativePlanQuestionCall) => { c.questions[1]!.options.push({ ...c.questions[1]!.options[0]! }); }, - (c: NativePlanQuestionCall) => { c.questions.push(structuredClone(issue().questions[0]!)); }, - (c: NativePlanQuestionCall) => { c.answeredAt = 'invalid'; }, - ]) { - const c = setupPacket(); mutate(c); expect(engStep0Boundary(fp(c))).toBe(false); - } - expect(engStep0Boundary({ ...fp(setupPacket()), signature: 'foreign:call' })).toBe(false); - expect(engStep0Boundary({ ...fp(setupPacket()), options: [] })).toBe(false); - const c = setupPacket(); - c.questions[1]!.header = 'Issue 1'; - expect(engStep0Boundary(fp(c))).toBe(false); - }); - - test('retry ordinary Issue identity starts review without qids, retaining its later TODO', () => { - const { counts, phases } = census(retryCalls); - expect(counts).toEqual({ setup: 3, review: 6, administrative: 0 }); - expect(phases.slice(3).every(p => !p.preReview && !p.administrative)).toBe(true); - for (const call of retryCalls.slice(3, 8)) expect(isFirst(call)).toBe(true); - expect(retryCalls[8]!.questions[0]!.question).toContain('TODO 1'); - expect(captured.cases.retry.actual.reviewCount).toBe(0); - }); - - test('ordinary issue presentation can vary while completed identity and section number remain bound', () => { - for (const title of ['Issue 1', 'Finding 1.2 (D17)', 'D42 — Issue 1']) { - const call = changeQuestion(issue(), s => s.replace('Issue 1 (D4)', title).replace('AuthCache', 'SessionCache')); - call.questions[0]!.header = title.includes('1.2') ? 'Architecture 1.2' : 'Architecture 1'; - call.questions[0]!.options.reverse(); - for (const option of call.questions[0]!.options) { - call.answers = { [call.questions[0]!.question]: option.label }; - expect(isFirst(call)).toBe(true); - } - } - }); - - test('setup, quoted or mismatched section identities do not start review', () => { - for (const header of ['Scope', 'Approach', 'Next steps', 'Arch 2', 'Example Arch 1', 'TODO 1']) { - const call = issue(); call.questions[0]!.header = header; expect(isFirst(call)).toBe(false); - } - for (const change of [ - (s: string) => '> ' + s, - (s: string) => 'Example: ' + s, - (s: string) => '```text\n' + s + '\n```', - (s: string) => s.replace('Issue 1 (D4)', 'Issue 2 (D4)'), - (s: string) => s + ' ', - ]) expect(isFirst(changeQuestion(issue(), change))).toBe(false); - for (const call of [...firstCalls.slice(0, 4), ...retryCalls.slice(0, 3)]) expect(isFirst(call)).toBe(false); - }); - - test('an Issue heading alone cannot turn a confirmation or report action into a finding', () => { - for (const body of [ - 'No defect remains in the cache. Proceed with the next section?', - 'The cache already serializes writes. Confirm this is accurate?', - 'Should I save the reviewed plan now?', - 'Add a section to the reviewed plan?', - 'Serialize the reviewed plan as JSON for the handoff?', - 'Add the completed tests to this report?', - ]) { - const c = changeQuestion(issue(), () => 'Issue 1 (D4) — ' + body); - c.questions[0]!.options = [ - { label: 'Yes', description: 'Confirm this statement; no new implementation work.' }, - { label: 'No', description: 'Do not confirm; no new implementation work.' }, - ]; - c.answers = { [c.questions[0]!.question]: 'Yes' }; - expect(isFirst(c)).toBe(false); - } - const c = issue(); - c.questions[0]!.options = [{ label: 'Yes', description: 'Confirm; no new work.' }, { label: 'No', description: 'Decline; no new work.' }]; - c.answers = { [c.questions[0]!.question]: 'Yes' }; - expect(isFirst(c)).toBe(false); - }); - - test('new first-finding path requires exact completed native identity and answer', () => { - for (const factory of [issue]) { - const classify = isFirst; - for (const mutate of [ - (c: NativePlanQuestionCall) => { c.answered = false; }, - (c: NativePlanQuestionCall) => { c.failed = true; }, - (c: NativePlanQuestionCall) => { delete c.failed; }, - (c: NativePlanQuestionCall) => { c.sessionId = ''; }, - (c: NativePlanQuestionCall) => { c.toolUseId = ''; }, - (c: NativePlanQuestionCall) => { delete c.answeredAt; }, - (c: NativePlanQuestionCall) => { c.answeredAt = 'invalid'; }, - (c: NativePlanQuestionCall) => { c.unansweredQuestionIndices = [0]; }, - (c: NativePlanQuestionCall) => { delete c.unansweredQuestionIndices; }, - (c: NativePlanQuestionCall) => { c.questions[0]!.multiSelect = true; }, - (c: NativePlanQuestionCall) => { c.questions.push(structuredClone(c.questions[0]!)); }, - (c: NativePlanQuestionCall) => { c.questions[0]!.options[1]!.label = c.questions[0]!.options[0]!.label; }, - (c: NativePlanQuestionCall) => { c.answers = {}; }, - (c: NativePlanQuestionCall) => { c.answers = { [c.questions[0]!.question]: 'Unoffered' }; }, - (c: NativePlanQuestionCall) => { c.answers!.foreign = 'Foreign'; }, - ]) { const c = factory(); mutate(c); expect(classify(c)).toBe(false); } - const classifyFp = engFirstReviewAUQ; - expect(classifyFp({ ...fp(factory()), signature: 'foreign:call' })).toBe(false); - expect(classifyFp({ ...fp(factory()), nativeCall: undefined })).toBe(false); - expect(classifyFp({ ...fp(factory()), nativeQuestionIndex: 1 })).toBe(false); - expect(classifyFp({ ...fp(factory()), options: [] })).toBe(false); - const wrong = fp(factory()); wrong.options[0]!.index = 2; expect(classifyFp(wrong)).toBe(false); - } - }); - - test('new evidence selects precisely its affected existing paid workflows', () => { - const selected = (path: string) => Object.entries(E2E_TOUCHFILES).filter(([, patterns]) => patterns.some(p => matchGlob(path, p))).map(([name]) => name).sort(); - for (const path of ['test/eng-count-ad-v2.test.ts', 'test/fixtures/eng-count-ad-v2.json']) { - expect(selected(path)).toEqual(['plan-eng-multi-finding-batching']); - } - }); -}); diff --git a/test/eng-declarative-as.test.ts b/test/eng-declarative-as.test.ts deleted file mode 100644 index 6e562c9f7..000000000 --- a/test/eng-declarative-as.test.ts +++ /dev/null @@ -1,126 +0,0 @@ -import { expect, test } from 'bun:test'; -import captured from './fixtures/eng-declarative-as.json'; -import { engFirstReviewAUQ, engSetupAUQ, engStep0Boundary, nativePlanCallFingerprint, planCountQuestionPhase } from './helpers/claude-pty-runner'; -import type { NativePlanQuestionCall } from './helpers/plan-count-transcript'; -import { E2E_TOUCHFILES, selectTests } from './helpers/touchfiles'; - -const actual = () => structuredClone(captured.call) as NativePlanQuestionCall; -function answered(c: NativePlanQuestionCall, index = 0) { - c.answers = { [c.questions[0]!.question]: c.questions[0]!.options[index]!.label }; - return nativePlanCallFingerprint(c, 0, true); -} -function edit(replace: (text: string) => string) { - const c = actual(), q = c.questions[0]!; - q.question = replace(q.question); - q.options.forEach(o => { o.label = replace(o.label); o.description = replace(o.description ?? ''); }); - return c; -} - -test('a completed declarative cache issue starts review without a question mark or qid', () => { - const fp = answered(actual()); - expect(engFirstReviewAUQ(fp)).toBe(true); - expect(engSetupAUQ(fp)).toBe(false); - expect(planCountQuestionPhase(fp, false, engStep0Boundary, engFirstReviewAUQ, engSetupAUQ)).toMatchObject({ preReview: false, reviewStarted: true }); - expect(captured.provenance.retrospectivePass).toBe(false); -}); - -test('all answers, option orders, identifiers and consistent ordinals qualify', () => { - for (const reverse of [false, true]) for (let i = 0; i < 3; i++) { - const c = actual(); if (reverse) c.questions[0]!.options.reverse(); - expect(engFirstReviewAUQ(answered(c, i))).toBe(true); - } - for (const name of ['TenantCache', '$Shared', '_Store']) expect(engFirstReviewAUQ(answered(edit(t => t.replaceAll('AuthCache', name))))).toBe(true); - for (const [one, two] of [['First', 'Second'], ['Z_store', '$Reader'], ['SessionMint', 'AuthBroker']]) { - expect(engFirstReviewAUQ(answered(edit(t => t.replaceAll('AuthBroker', '__first__').replaceAll('SessionMint', two).replaceAll('__first__', one))))).toBe(true); - } - for (const kind of ['Issue', 'Finding']) expect(engFirstReviewAUQ(answered(edit(t => t.replace('Issue 1 ', `${kind} 17 `).replace(/\b1([A-C])\b/g, '17$1'))))).toBe(true); -}); - -test('native completion and matching answered menu remain required', () => { - const mutations: Array<(c: NativePlanQuestionCall) => void> = [ - c => { c.answered = false; }, c => { c.failed = true; }, c => { c.answers = {}; }, - c => { c.answers = { [c.questions[0]!.question]: 'unoffered' }; }, c => { c.answers!['foreign'] = 'answer'; }, - c => { c.answeredAt = 'invalid'; }, c => { c.unansweredQuestionIndices = [0]; }, - c => { c.sessionId = ''; }, c => { c.toolUseId = ''; }, c => { c.questions[0]!.multiSelect = true; }, - c => { c.questions.push(structuredClone(c.questions[0]!)); }, - c => { c.questions[0]!.options[1]!.label = c.questions[0]!.options[0]!.label; }, - ]; - for (const mutate of mutations) { const c = actual(); mutate(c); expect(engFirstReviewAUQ(nativePlanCallFingerprint(c, 0, true))).toBe(false); } - const fp = answered(actual()); - expect(engFirstReviewAUQ({ ...fp, signature: 'foreign' })).toBe(false); - expect(engFirstReviewAUQ({ ...fp, nativeQuestionIndex: 1 })).toBe(false); - expect(engFirstReviewAUQ({ ...fp, options: [...fp.options].reverse() })).toBe(false); -}); - -test('the brief must own a current architecture assessment and distinct writers', () => { - const changes = [ - (t: string) => 'Example: ' + t, (t: string) => '> ' + t, (t: string) => '```\n' + t + '\n```', - (t: string) => t.replace('Section 1 Architecture', 'Section 1 Administration'), - (t: string) => t.replace('ELI10: Two services', 'ELI10: If two services'), - (t: string) => t.replace('ELI10: Two services', 'ELI10: Source example: Two services'), - (t: string) => t.replace('AuthBroker and SessionMint share', 'AuthBroker and AuthBroker share'), - (t: string) => t.replace('share a global mutable', 'used to share a global mutable'), - (t: string) => t.replace('writes can interleave', 'writes are serialized').replace('no ordering', 'per-key ordering'), - (t: string) => t.replace('Project/branch/task:', 'Historical assessment:'), - ]; - for (const change of changes) { const c = actual(); c.questions[0]!.question = change(c.questions[0]!.question); expect(engFirstReviewAUQ(answered(c))).toBe(false); } - for (const header of ['Setup', 'TODOs', 'Issue 2', 'Review report']) { const c = actual(); c.questions[0]!.header = header; expect(engFirstReviewAUQ(answered(c))).toBe(false); } -}); - -test('same-decision withdrawals and contrary current state invalidate the issue', () => { - for (const status of ['withdrawn', 'superseded', 'rejected', 'cancelled', 'resolved', 'closed', 'not current', 'no longer current']) { - for (const literal of [status, `"${status}"`, `“${status}”`, `'${status}'`, `‘${status}’`, '`' + status + '`']) for (const target of ['question', 'remedy', 'unchanged']) { - const c = actual(), q = c.questions[0]!, suffix = ` This finding is ${literal}.`; - if (target === 'question') q.question += suffix; else q.options[target === 'remedy' ? 0 : 2]!.description += suffix; - expect(engFirstReviewAUQ(answered(c))).toBe(false); - } - } - for (const contradiction of ['The cache is no longer global.', 'The services no longer mutate shared state.', 'No current risk remains.']) { - const c = actual(); c.questions[0]!.question += '\n' + contradiction; expect(engFirstReviewAUQ(answered(c))).toBe(false); - } -}); - -test('technical options cannot be quoted, hypothetical, mismatched or cancelled', () => { - for (const index of [0, 2]) for (const frame of ['Example: ', 'If approved: ', 'Source excerpt: ', '> ']) { - const c = actual(); c.questions[0]!.options[index]!.description = frame + c.questions[0]!.options[index]!.description; - expect(engFirstReviewAUQ(answered(c))).toBe(false); - } - for (const [index, suffix] of [[0, ' Correction: Do not remove the module-level export.'], [0, ' The module export remains.'], [0, ' Writes remain unordered.'], [2, ' Correction: Do not proceed as written.'], [2, ' The race is resolved.']] as const) { - const c = actual(); c.questions[0]!.options[index]!.description += suffix; expect(engFirstReviewAUQ(answered(c))).toBe(false); - } - for (const index of [0, 2]) { const c = actual(); c.questions[0]!.options[index]!.label = '1' + (index ? 'C' : 'A') + ') Record in report'; expect(engFirstReviewAUQ(answered(c))).toBe(false); } - const c = actual(); c.questions[0]!.options[0]!.description = c.questions[0]!.options[0]!.description!.replaceAll('AuthCache', 'UnrelatedCache'); - expect(engFirstReviewAUQ(answered(c))).toBe(false); -}); - -test('new boundary regressions select the two Eng count owners', () => { - for (const file of ['test/eng-declarative-as.test.ts', 'test/fixtures/eng-declarative-as.json']) expect(selectTests([file], E2E_TOUCHFILES, []).selected.sort()).toEqual(['plan-eng-multi-finding-batching']); -}); - - -test('current named-owner contradictions are distinct from foreign and archived references', () => { - for (const statement of ['AuthCache is no longer global.', 'AuthCache is no longer mutable.', 'AuthBroker no longer mutates the cache.', 'SessionMint no longer writes to the cache.', 'D4 is withdrawn.', 'Correction: This finding is withdrawn.', 'Correction: Issue 1 is "withdrawn".', 'Correction: AuthCache is no longer global.', "Issue 1 is 'withdrawn'.", 'This finding is “withdrawn”.']) { - for (const boundary of ['\n', '; ']) { - const c = actual(); c.questions[0]!.question += boundary + statement; - expect(engFirstReviewAUQ(answered(c))).toBe(false); - } - } - for (const statement of ['Issue 19 is withdrawn.', 'D42 is withdrawn.', 'AnotherCache is no longer global.', 'An archived review recorded this finding is "withdrawn".', "An archived review recorded this finding is 'withdrawn'.", 'The prior report said "This finding is withdrawn."', '> This finding is withdrawn.']) { - const c = actual(); c.questions[0]!.question += '\n' + statement; - expect(engFirstReviewAUQ(answered(c))).toBe(true); - } - for (const replacement of ['an unrelated billing cache', 'a different cache', 'an OtherCache']) { - const c = actual(); c.questions[0]!.options[2]!.description = c.questions[0]!.options[2]!.description!.replace('an auth cache', replacement); - expect(engFirstReviewAUQ(answered(c))).toBe(false); - } - const c = actual(); c.questions[0]!.options[2]!.description = c.questions[0]!.options[2]!.description!.replace('an auth cache', 'an AuthCache'); - expect(engFirstReviewAUQ(answered(c))).toBe(true); -}); - - -test('the unchanged option cannot contradict its own remaining cache risk', () => { - for (const statement of ['Correction: The writers are now serialized.', 'AuthCache is no longer global.', 'The cache is removed.']) { - const c = actual(); c.questions[0]!.options[2]!.description += '\n' + statement; - expect(engFirstReviewAUQ(answered(c))).toBe(false); - } -}); diff --git a/test/eng-declared-retry-at.test.ts b/test/eng-declared-retry-at.test.ts deleted file mode 100644 index 54f71e9e5..000000000 --- a/test/eng-declared-retry-at.test.ts +++ /dev/null @@ -1,82 +0,0 @@ -import {describe, expect, test} from 'bun:test'; -import {engFirstReviewAUQ, engSetupAUQ, nativePlanCallFingerprint} from './helpers/claude-pty-runner'; -import type {NativePlanQuestionCall} from './helpers/plan-count-transcript'; -import {selectTests} from './helpers/touchfiles'; -import {E2E_TOUCHFILES} from './helpers/touchfiles-data'; -import captured from './fixtures/eng-declared-retry-at.json'; -const first=()=>structuredClone(captured) as NativePlanQuestionCall; -const fp=(c=first())=>nativePlanCallFingerprint(c,0,true); -const classify=(c=first())=>engFirstReviewAUQ(fp(c)); -function mutated(fn:(c:NativePlanQuestionCall)=>void){const c=first();fn(c);return c;} -function text(fn:(s:string)=>string){return mutated(c=>{const q=c.questions[0]!,answer=c.answers![q.question]!;q.question=fn(q.question);c.answers={[q.question]:answer};});} -describe('declarative engineering retry choice',()=>{ - test('recognizes the actual answered finding without requiring a question mark',()=>{ - expect(classify()).toBe(true);expect(engSetupAUQ(fp())).toBe(false); - }); - test('consistent issue numbers, option order and chosen option may vary',()=>{ - const c=first(),q=c.questions[0]!;q.question=q.question.replace('D2 — Issue 1:','D8 — Issue 4:').replace('Recommendation: 1A','Recommendation: 4A');q.header='Issue 4';q.options.forEach(o=>{o.label=o.label.replace(/^1/,'4');});q.options.reverse(); - for(const o of q.options){c.answers={[q.question]:o.label};expect(classify(c)).toBe(true);} - }); - test('native completion, original menu and response ownership remain mandatory',()=>{ - for(const change of [ - (c:NativePlanQuestionCall)=>{c.answered=false;},(c:NativePlanQuestionCall)=>{c.failed=true;},(c:NativePlanQuestionCall)=>{delete c.failed;},(c:NativePlanQuestionCall)=>{c.answeredAt='invalid';},(c:NativePlanQuestionCall)=>{c.sessionId='';},(c:NativePlanQuestionCall)=>{c.toolUseId='';},(c:NativePlanQuestionCall)=>{c.answers={};},(c:NativePlanQuestionCall)=>{c.answers={[c.questions[0]!.question]:'unoffered'};},(c:NativePlanQuestionCall)=>{c.unansweredQuestionIndices=[0];},(c:NativePlanQuestionCall)=>{c.questions.push(structuredClone(c.questions[0]!));},(c:NativePlanQuestionCall)=>{c.questions[0]!.multiSelect=true;}, - ])expect(classify(mutated(change))).toBe(false); - for(const f of [{...fp(),signature:'foreign:call'},{...fp(),nativeQuestionIndex:1},{...fp(),options:[...fp().options].reverse()}])expect(engFirstReviewAUQ(f)).toBe(false); - }); - test('issue identity and report administration cannot substitute for the finding',()=>{ - for(const [a,b] of [['Issue 1:','Issue 0:'],['Issue 1:','Issue 01:'],['Issue 1:','Finding 1:'],['PLAN.md:6-8','PLAN.md:0-8'],['D2 —','D02 —']])expect(classify(text(s=>s.replace(a!,b!)))).toBe(false); - for(const h of ['Issue 2','Routing','Report','Scope'])expect(classify(mutated(c=>{c.questions[0]!.header=h;}))).toBe(false); - expect(classify(text(s=>s.replace(/^Project\/branch\/task:.*\n/m,'')))).toBe(false); - expect(classify(text(s=>s.replace('ELI10:','Project/branch/task: unrelated\nELI10:')))).toBe(false); - expect(classify(mutated(c=>{c.questions[0]!.options[0]!.label='1A) Continue the review';c.answers={[c.questions[0]!.question]:c.questions[0]!.options[0]!.label};}))).toBe(false); - }); - test('quoted, hypothetical and closed current findings remain excluded',()=>{ - for(const prefix of ['Source excerpt: ','Earlier review assessment: ','If approved, ','Provided approval, ']){ - expect(classify(text(s=>s.replace('ELI10: ','ELI10: '+prefix)))).toBe(false); - for(const i of [0,2])expect(classify(mutated(c=>{c.questions[0]!.options[i]!.description=prefix+c.questions[0]!.options[i]!.description;}))).toBe(false); - } - for(const status of ['withdrawn','resolved','"closed"','“superseded”']){ - expect(classify(text(s=>s+` This finding is ${status}.`))).toBe(false); - for(const i of [0,2])expect(classify(mutated(c=>{c.questions[0]!.options[i]!.description+=` This option is ${status}.`;}))).toBe(false); - } - expect(classify(text(s=>s+'\n"Earlier review assessment: This finding is withdrawn."'))).toBe(true); - expect(classify(text(s=>s+'\nCorrection: retry scheduling no longer runs inside each worker.'))).toBe(false); - }); - test('remedy and unchanged choice each retain their own current consequence',()=>{ - for(const [i,a,b] of [[0,'come from the library','might be evaluated later'],[0,'a pure function, trivially unit-tested','five separate implementations'],[2,'a crash or deploy mid-backoff drops the retry','a crash or deploy preserves every retry']] as const)expect(classify(mutated(c=>{const o=c.questions[0]!.options[i]!;o.description=o.description!.replace(a,b);}))).toBe(false); - expect(classify(mutated(c=>{c.questions[0]!.options[0]!.description+='\nCorrection: the library will not own persistence.';}))).toBe(false); - expect(classify(mutated(c=>{c.questions[0]!.options[0]!.description+='\nCorrection: do not use the library retry hook.';}))).toBe(false); - expect(classify(mutated(c=>{c.questions[0]!.options[2]!.description+='\nCorrection: the per-worker scheduler is now crash-safe.';}))).toBe(false); - }); - test('fixture changes select only their workflow',()=>{ - for(const f of ['test/eng-declared-retry-at.test.ts','test/fixtures/eng-declared-retry-at.json'])expect(selectTests([f],E2E_TOUCHFILES,[]).selected).toEqual(['plan-eng-multi-finding-batching']); - }); -}); - -describe('current owner status and approval boundaries',()=>{ -const ownedStatusCases:Array<{name:string,expected:boolean,edit:(c:any)=>void}>=[];const add=(name:string,expected:boolean,edit:(c:any)=>void)=>ownedStatusCases.push({name,expected,edit}); -const question=(c:any,suffix:string)=>{const q=c.questions[0],answer=c.answers[q.question];q.question+=suffix;c.answers={[q.question]:answer}}; -add('exact completed declarative choice',true,()=>{}); -for(const owner of ['This finding','Issue 1','D2'])for(const status of ['withdrawn','not current','no longer current'])for(const quote of ['',"'",'‘'])add(`current ${owner} ${quote}${status}`,false,c=>question(c,`\n${owner} is ${quote}${status}${quote==='‘'?'’':quote}.`)); -for(const i of [0,2])for(const status of ['withdrawn','not current','no longer current'])for(const quote of ['',"'",'‘'])add(`option ${i} ${quote}${status}`,false,c=>{c.questions[0].options[i].description+=`\nThis option is ${quote}${status}${quote==='‘'?'’':quote}.`}); -for(const i of [0,2])for(const condition of ['This option applies only if approved.','This option is conditional on approval.','If approved, proceed with this option.'])add(`option ${i} condition ${condition}`,false,c=>{c.questions[0].options[i].description+='\n'+condition}); -for(const condition of ['This finding applies only if approved.','This finding is conditional on approval.'])add('finding condition '+condition,false,c=>question(c,'\n'+condition)); -for(const owner of ['Issue 2','D3'])add('foreign closed owner '+owner,true,c=>question(c,`\n${owner} is withdrawn.`)); -for(const i of [0,2])add(`quoted historical option${i}`,true,c=>{c.questions[0].options[i].description+='\nEarlier review assessment: "This option is withdrawn."';}); -add('quoted historical finding',true,c=>question(c,'\n"Earlier review assessment: This finding is withdrawn."')); -add('quoted title',false,c=>{const q=c.questions[0],a=c.answers[q.question];q.question=q.question.replace(/^(.*)\n/,'"$1"\n');c.answers={[q.question]:a}}); -add('absent completion',false,c=>{c.answered=false});add('failed native result',false,c=>{c.failed=true});add('invalid answer time',false,c=>{c.answeredAt='missing'}); -add('remedy and unchanged outcomes reversed',false,c=>{const o=c.questions[0].options;[o[0].description,o[2].description]=[o[2].description,o[0].description]}); -add('remedy actually declines library persistence',false,c=>{c.questions[0].options[0].description+='\nThe library will not own persistence.'}); -add('unchanged is now crash safe',false,c=>{c.questions[0].options[2].description+='\nThe scheduler is now crash-safe.'}); -for(const control of ownedStatusCases)test(control.name,()=>{const call=first();control.edit(call);expect(classify(call)).toBe(control.expected);}); -}); - - test('bold current owners keep their scalar status before source quotes are removed',()=>{ - for(const owner of ['This finding','D2']){ - expect(classify(text(s=>s+`\n**${owner}** is 'withdrawn'.`))).toBe(false); - expect(classify(text(s=>s+`\n**${owner}** is ‘withdrawn’.`))).toBe(false); - } - for(const i of [0,2])expect(classify(mutated(c=>{c.questions[0]!.options[i]!.description+=`\n**This option** is 'withdrawn'.`;}))).toBe(false); - expect(classify(text(s=>s+'\n"Earlier review assessment: **This finding** is withdrawn."'))).toBe(true); - }); diff --git a/test/eng-first-category-af.test.ts b/test/eng-first-category-af.test.ts deleted file mode 100644 index 2abe565a6..000000000 --- a/test/eng-first-category-af.test.ts +++ /dev/null @@ -1,103 +0,0 @@ -import { expect, test } from 'bun:test'; -import captured from './fixtures/eng-first-category-af.json'; -import { engFirstReviewAUQ, engSetupAUQ, engStep0Boundary, nativePlanCallFingerprint, planCountQuestionPhase } from './helpers/claude-pty-runner'; -import type { NativePlanQuestionCall } from './helpers/plan-count-transcript'; -import { E2E_TOUCHFILES, selectTests } from './helpers/touchfiles'; - -const actual = () => structuredClone(captured.fingerprint.nativeCall) as NativePlanQuestionCall; - -test('actual completed Architecture issue starts review', () => { - const fp = nativePlanCallFingerprint(actual(), 0, true); - expect(engFirstReviewAUQ(fp)).toBe(true); - expect(engSetupAUQ(fp)).toBe(false); - expect(planCountQuestionPhase(fp, false, engStep0Boundary, engFirstReviewAUQ, engSetupAUQ)) - .toMatchObject({ preReview: false, reviewStarted: true }); - expect(captured.provenance.retrospectivePass).toBe(false); -}); - -function answer(c: NativePlanQuestionCall, index = 0) { - c.answers = {[c.questions[0]!.question]: c.questions[0]!.options[index]!.label}; - return nativePlanCallFingerprint(c, 0, true); -} - -test('all offered choices and menu orders remain substantive decisions', () => { - for (const reverse of [false, true]) for (let index = 0; index < 3; index++) { - const c = actual(); if (reverse) c.questions[0]!.options.reverse(); - expect(engFirstReviewAUQ(answer(c, index))).toBe(true); - } -}); - -test('identifier spelling, writer order and matching issue numbers are incidental', () => { - for (const [left, right] of [['TenantReader', 'SessionWriter'], ['Z_store', '$AStore'], ['SessionMint', 'AuthBroker']]) { - const c = actual(); const q = c.questions[0]!; - q.question = q.question.replace('AuthBroker and SessionMint', `${left} and ${right}`); - expect(engFirstReviewAUQ(answer(c))).toBe(true); - } - for (const kind of ['Issue', 'Finding']) { - const c = actual(); c.questions[0]!.question = c.questions[0]!.question.replace('D4 — Issue 1', `D87 — ${kind} 12.3`); - c.questions[0]!.header = `${kind} 12.3`; - expect(engFirstReviewAUQ(answer(c))).toBe(true); - } -}); - -test('native completion, timestamp, answer and menu identity remain mandatory', () => { - const mutations: Array<(c: NativePlanQuestionCall) => void> = [ - c => { c.answered = false; }, c => { c.failed = true; }, c => { c.answers = {}; }, - c => { c.answers = {[c.questions[0]!.question]: 'not offered'}; }, - c => { c.questions[0]!.question += ' changed'; }, - c => { c.unansweredQuestionIndices = [0]; }, c => { c.answeredAt = 'invalid'; }, - c => { c.sessionId = ''; }, c => { c.toolUseId = ''; }, - c => { c.questions[0]!.multiSelect = true; }, - c => { c.questions.push(structuredClone(c.questions[0]!)); }, - c => { c.questions[0]!.options[1]!.label = c.questions[0]!.options[0]!.label; }, - ]; - for (const mutate of mutations) { - const c = actual(); mutate(c); expect(engFirstReviewAUQ(nativePlanCallFingerprint(c, 0, true))).toBe(false); - } - const fp = nativePlanCallFingerprint(actual(), 0, true); - expect(engFirstReviewAUQ({...fp,signature:'foreign'})).toBe(false); - expect(engFirstReviewAUQ({...fp,nativeQuestionIndex:1})).toBe(false); - expect(engFirstReviewAUQ({...fp,options:[...fp.options].reverse()})).toBe(false); -}); - -test('administrative, TODO, uncertain and quoted contexts cannot borrow technical labels', () => { - const base = actual().questions[0]!.question.split('\n')[0]!; - const titles = [ - 'D4 — Issue 1 (Architecture): Record the completed review in TODOs?', - 'D4 — Issue 1 (Architecture): Confirm that the shared cache review is complete?', - 'D4 — Issue 1 (Architecture): Which review runs next?', - base.replace('AuthBroker and SessionMint both mutate', 'If AuthBroker and SessionMint both mutate'), - base.replace('AuthBroker and SessionMint', 'AuthBroker and AuthBroker'), - base.replace('with no owner and no serialization', 'with an owner and per-key serialization'), - 'Example: ' + base, '> ' + base, '```\n' + base, - base.replace('How should shared-state access be structured?', 'Should the review report record this finding?'), - ]; - for (const title of titles) { - const c = actual(); c.questions[0]!.question = title; - expect(engFirstReviewAUQ(answer(c))).toBe(false); - } - for (const header of ['Issue 2', 'Issue 1.2', 'TODOs', 'Setup', 'Next review']) { - const c = actual(); c.questions[0]!.header = header; - expect(engFirstReviewAUQ(answer(c))).toBe(false); - } -}); - -test('opposed implementation choices cannot be replaced by report or workflow choices', () => { - for (const labels of [ - ['Record in report', 'Defer the report', 'Keep the report'], - ['Run Eng next', 'Run Design next', 'Keep reviewing manually'], - ]) { - const c = actual(); c.questions[0]!.options.forEach((o, i) => {o.label = labels[i]!;}); - expect(engFirstReviewAUQ(answer(c))).toBe(false); - } - const c = actual(); c.questions[0]!.options[0]!.description = ''; - expect(engFirstReviewAUQ(answer(c))).toBe(false); -}); - -test('regression evidence selects only the two affected Eng count owners', () => { - for (const file of ['test/eng-first-category-af.test.ts', 'test/fixtures/eng-first-category-af.json']) { - const owners = Object.entries(E2E_TOUCHFILES).filter(([, files]) => files.includes(file)).map(([owner])=>owner).sort(); - expect(owners).toEqual(['plan-eng-multi-finding-batching']); - expect(selectTests([file], E2E_TOUCHFILES, []).selected.sort()).toEqual(owners); - } -}); diff --git a/test/eng-first-review-t.test.ts b/test/eng-first-review-t.test.ts deleted file mode 100644 index 3dac8c85f..000000000 --- a/test/eng-first-review-t.test.ts +++ /dev/null @@ -1,81 +0,0 @@ -import { describe, expect, test } from 'bun:test'; -import { readFileSync } from 'node:fs'; -import { join } from 'node:path'; -import { nativePlanCallFingerprint, planCountQuestionPhase, engStep0Boundary, engSetupAUQ, engFirstReviewAUQ } from './helpers/claude-pty-runner'; -import type { NativePlanQuestionCall } from './helpers/plan-count-transcript'; - -const calls: NativePlanQuestionCall[] = JSON.parse(readFileSync(join(import.meta.dir, 'fixtures/eng-batching-t-calls.json'), 'utf8')); -const fresh = () => structuredClone(calls[0]!); -const first = (call: NativePlanQuestionCall) => engFirstReviewAUQ(nativePlanCallFingerprint(call, 0, true)); -function question(call: NativePlanQuestionCall, text: string) { - const q = call.questions[0]!; const answer = call.answers![q.question]; - call.answers = { [text]: answer! }; q.question = text; -} - -describe('T Eng first architecture choice', () => { - test('the actual first architecture issue starts review on this call', () => { - const call = fresh(); const fp = nativePlanCallFingerprint(call, 0, true); - expect(engSetupAUQ(fp)).toBe(false); - expect(first(call)).toBe(true); - expect(planCountQuestionPhase(fp, false, engStep0Boundary, engFirstReviewAUQ, engSetupAUQ)) - .toEqual({ preReview: false, reviewStarted: true }); - }); - - test('all five exact native decisions remain separate review calls', () => { - let started = false; - const phases = calls.map(call => { - const fp = nativePlanCallFingerprint(call, 0, !started); - const phase = planCountQuestionPhase(fp, started, engStep0Boundary, engFirstReviewAUQ, engSetupAUQ); - started = phase.reviewStarted; return phase.preReview; - }); - expect(phases).toEqual([false, false, false, false, false]); - }); - - test('offered answer identity survives reorder and either alternative decision', () => { - const call = fresh(); call.questions[0]!.options.reverse(); - expect(first(call)).toBe(true); - for (const option of call.questions[0]!.options) { - call.answers![call.questions[0]!.question] = option.label; - expect(first(call)).toBe(true); - } - }); - - test('requires one completed native question and exact offered answer', () => { - const variants: Array<(c: NativePlanQuestionCall) => void> = [ - c => { c.answered = false; }, c => { c.failed = true; }, - c => { delete c.unansweredQuestionIndices; }, c => { c.unansweredQuestionIndices = [0]; }, - c => { c.answers = {}; }, c => { c.answers![c.questions[0]!.question] = 'Foreign answer'; }, - c => { c.questions.push(structuredClone(c.questions[0]!)); }, - c => { c.questions[0]!.multiSelect = true; }, - c => { c.questions[0]!.options[1]!.label = c.questions[0]!.options[0]!.label; }, - ]; - for (const change of variants) { const call = fresh(); change(call); expect(first(call)).toBe(false); } - const fp = nativePlanCallFingerprint(fresh(), 0, true); fp.signature = 'foreign:identity'; - expect(engFirstReviewAUQ(fp)).toBe(false); - delete fp.nativeCall; expect(engFirstReviewAUQ(fp)).toBe(false); - }); - - test('whole-plan approach, setup, missing identity and quoted examples cannot start review', () => { - for (const header of ['Approach', 'Scope', 'Routing rules', 'Next review']) { - const call = fresh(); call.questions[0]!.header = header; expect(first(call)).toBe(false); - } - const original = fresh().questions[0]!.question; - for (const text of [ - original.replace('arch-retry-scheduler', 'arch-setup'), original.replace(/ ]+>/, ''), - original.replace('Architecture: Custom retry scheduler', 'Approach: Which whole-plan direction'), - '> ' + original, '```text\n' + original + '\n```', original + ' Should we start another review?', - ]) { const call = fresh(); question(call, text); expect(first(call)).toBe(false); } - }); - - test('requires an affirmative existing defect, not a neutral or negated comparison', () => { - for (const description of [ - 'Both implementations are equally valid choices.', - 'Each worker gets its own copy. There is no DRY violation.', - fresh().questions[0]!.options[2]!.description!.replace('acknowledged DRY violation', 'no DRY violation'), - fresh().questions[0]!.options[2]!.description!.replace('Creates 5 divergence points', 'No longer creates 5 divergence points'), - '```text\n' + fresh().questions[0]!.options[2]!.description + '\n```', - ]) { const call = fresh(); call.questions[0]!.options[2]!.description = description; expect(first(call)).toBe(false); } - const call = fresh(); call.questions[0]!.options[0]!.label = 'Run /office-hours'; - call.answers![call.questions[0]!.question] = 'Run /office-hours'; expect(first(call)).toBe(false); - }); -}); diff --git a/test/eng-first-review.test.ts b/test/eng-first-review.test.ts new file mode 100644 index 000000000..26d19c8bc --- /dev/null +++ b/test/eng-first-review.test.ts @@ -0,0 +1,1469 @@ +/** + * Eng first-review classification (engStep0Boundary / engSetupAUQ / engFirstReviewAUQ) over captured native calls. + */ +import { describe } from 'bun:test'; +import { expect } from 'bun:test'; +import { test } from 'bun:test'; +import captured_eng_annotated_cache_au from './fixtures/eng-annotated-cache-au.json'; +import { engFirstReviewAUQ } from './helpers/claude-pty-runner'; +import { engSetupAUQ } from './helpers/claude-pty-runner'; +import { engStep0Boundary } from './helpers/claude-pty-runner'; +import { nativePlanCallFingerprint } from './helpers/claude-pty-runner'; +import { planCountQuestionPhase } from './helpers/claude-pty-runner'; +import type { NativePlanQuestionCall } from './helpers/plan-count-transcript'; +import fixture_eng_architecture_cache_av from './fixtures/eng-architecture-cache-av-calls.json'; +import captured_eng_binding_retry_z from './fixtures/eng-binding-retry-z-calls.json'; +import captured_eng_binding_z from './fixtures/eng-binding-z-calls.json'; +import type { AskUserQuestionFingerprint as FP } from './helpers/claude-pty-runner'; +import fixture_eng_cache_brief_am from './fixtures/eng-cache-brief-am.json'; +import fixture_eng_cache_owner_an from './fixtures/eng-cache-owner-an.json'; +import type { AskUserQuestionFingerprint as Fingerprint } from './helpers/claude-pty-runner'; +import captured_eng_cache_writes_as from './fixtures/eng-cache-writes-as.json'; +import captured_eng_count_ad_v2 from './fixtures/eng-count-ad-v2.json'; +import captured_eng_declarative_as from './fixtures/eng-declarative-as.json'; +import captured_eng_declared_retry_at from './fixtures/eng-declared-retry-at.json'; +import captured_eng_first_category_af from './fixtures/eng-first-category-af.json'; +import { readFileSync } from 'node:fs'; +import { join } from 'node:path'; +import fixture_eng_injected_export_aq from './fixtures/eng-injected-export-aq.json'; +import fixture_eng_library_hooks_aq from './fixtures/eng-library-hooks-aq.json'; +import captured_eng_scope_y from './fixtures/eng-scope-y-calls.json'; + +describe('eng-annotated-cache-au', () => { +const captured = captured_eng_annotated_cache_au; +const fresh=()=>structuredClone(captured.call) as NativePlanQuestionCall; +const fp=(c=fresh())=>nativePlanCallFingerprint(c,Date.parse(c.answeredAt!),true); +const first=(c=fresh())=>engFirstReviewAUQ(fp(c)); +function change(edit:(q:NativePlanQuestionCall['questions'][number])=>void){const c=fresh(),q=c.questions[0]!,picked=q.options.findIndex(o=>o.label===c.answers[q.question]);edit(q);c.answers={[q.question]:q.options[picked]!.label};return c;} +test('exact acknowledged annotated cache finding opens review without changing native ownership',()=>{ + const c=fresh(),before=JSON.stringify(c);expect(first(c)).toBe(true);expect(engSetupAUQ(fp(c))).toBe(false);expect(planCountQuestionPhase(fp(c),false,engStep0Boundary,engFirstReviewAUQ,engSetupAUQ)).toMatchObject({preReview:false,reviewStarted:true});expect(JSON.stringify(c)).toBe(before);expect(captured.provenance.retrospectivePass).toBe(false); +}); +test('incidental metadata and all offered choices retain substantive review identity',()=>{ + for(const edit of [ + (q:any)=>{q.question=q.question.replace('D5 — Issue 1','D15 — Issue 11');q.header='Arch 11';}, + (q:any)=>{q.question=q.question.replace('PLAN.md:19-20 + :10','docs/plan.md:42');}, + (q:any)=>{q.question=q.question.replace('[P1] (confidence 8/10)','[P2] (confidence 10/10)');}, + (q:any)=>{q.question=q.question.replaceAll('AuthCache','TenantStore').replaceAll('SessionMint','SessionWriter').replaceAll('AuthBroker','AuthReader');q.options=q.options.map((o:any)=>({...o,description:o.description.replaceAll('AuthCache','TenantStore').replaceAll('SessionMint','SessionWriter').replaceAll('AuthBroker','AuthReader')}));}, + (q:any)=>{q.question+='\n"Historical note: This finding is withdrawn."';}, + (q:any)=>{q.options[0].description+='\n"This option is withdrawn."';}, + ])expect(first(change(edit))).toBe(true); + for(const reversed of [false,true])for(let i=0;i<3;i++){const c=fresh(),q=c.questions[0]!;if(reversed)q.options.reverse();c.answers={[q.question]:q.options[i]!.label};expect(first(c)).toBe(true);} +}); +const changes:Array<[string,(q:NativePlanQuestionCall['questions'][number])=>void]>=[ + ['foreign issue header',q=>{q.header='Arch 2';}],['missing issue',q=>{q.question=q.question.replace('Issue 1 ','');}],['missing annotation',q=>{q.question=q.question.replace('[P1] (confidence 8/10) ','');}],['missing source location',q=>{q.question=q.question.replace('PLAN.md:19-20 + :10 — ','');}],['invalid confidence',q=>{q.question=q.question.replace('confidence 8/10','confidence 11/10');}], + ['conditional defect',q=>{q.question=q.question.replace('both mutate','might both mutate');}],['same actor twice',q=>{q.question=q.question.replace('AuthBroker and SessionMint','AuthBroker and AuthBroker');}],['serialized title',q=>{q.question=q.question.replace('does not serialize mutations','serializes mutations');}], + ['source title',q=>{q.question='Source: '+q.question;}],['quoted title',q=>{const lines=q.question.split('\n');lines[0]='"'+lines[0]+'"';q.question=lines.join('\n');}],['source context',q=>{q.question=q.question.replace('Project/branch/task:','Source:');}],['historical context',q=>{q.question=q.question.replace('Project/branch/task:','Project/branch/task: Historical assessment:');}], + ['no own explanation',q=>{q.question=q.question.replace(/^ELI10:.*$/m,'');}],['quoted explanation',q=>{q.question=q.question.replace(/^ELI10: (.*)$/m,'ELI10: "$1"');}],['competing explanation',q=>{q.question+='\nELI10: There is no race.';}],['hypothetical explanation',q=>{q.question=q.question.replace('ELI10:','ELI10: If approved,');}],['missing race consequence',q=>{q.question=q.question.replace('the mint can land after the invalidation and a suspended tenant keeps a live session','the tenant always loses the session');}], + ['repair wrong cache',q=>{q.options[0]!.description=q.options[0]!.description!.replace('AuthCache passed','OtherCache passed');}],['same writer and reader',q=>{q.options[0]!.description=q.options[0]!.description!.replace('AuthBroker reads','SessionMint reads');}],['missing invalidation rejection',q=>{q.options[0]!.description=q.options[0]!.description!.replace('are rejected if the entry was invalidated since read','are accepted even when invalidated');}],['missing owned repair',q=>{q.options[0]!.description='Choose later.';}],['missing opposed risk',q=>{q.options[2]!.description='The race is closed.';}],['opposition now serialized',q=>{q.options[2]!.description+='\nThe writers are now serialized.';}],['reader also writes',q=>{q.options[0]!.description+='\nAuthBroker also writes.';}], +]; +test.each(changes)('%s cannot open review',(_,edit)=>expect(first(change(edit))).toBe(false)); +test('current statuses, framing and conditional approval are enforced on finding and offered outcomes',()=>{ + for(const status of ['withdrawn','no longer current','hypothetical','optional'])for(const [open,close]of [['',''],['"','"'],["'","'"],['“','”'],['‘','’'],['`','`']]){ + for(const owner of ['This finding','D5','Issue 1'])expect(first(change(q=>{q.question+=`\n**${owner}** is ${open}${status}${close}.`;})),`${owner} ${open}${status}`).toBe(false); + for(const i of [0,1,2])expect(first(change(q=>{q.options[i]!.description+=`\n**This option** is ${open}${status}${close}.`;}))).toBe(false); + } + for(const prefix of ['Source:','Historical assessment:','If approved,','Once approved,','Pending approval:'])for(const i of [0,1,2])expect(first(change(q=>{q.options[i]!.description=prefix+'\n'+q.options[i]!.description;})),prefix).toBe(false); +}); +test('native completion, timestamp, exact answer, session and visible menu stay mandatory',()=>{ + const edits:Array<(c:NativePlanQuestionCall)=>void>=[c=>{c.answered=false;},c=>{c.failed=true;},c=>{c.answers={};},c=>{c.answers[c.questions[0]!.question]='not offered';},c=>{c.unansweredQuestionIndices=[0];},c=>{c.answeredAt='invalid';},c=>{c.sessionId='';},c=>{c.toolUseId='';},c=>{c.questions[0]!.multiSelect=true;},c=>{c.questions.push(structuredClone(c.questions[0]!));},c=>{c.questions[0]!.options[1]!.label=c.questions[0]!.options[0]!.label;}]; + for(const edit of edits){const c=fresh();edit(c);expect(first(c)).toBe(false);}const f=fp();expect(engFirstReviewAUQ({...f,signature:'foreign'})).toBe(false);expect(engFirstReviewAUQ({...f,options:f.options.slice().reverse()})).toBe(false);expect(engFirstReviewAUQ({...f,nativeQuestionIndex:1})).toBe(false); +}); +test('current approval conditions and same-option effort boundaries cannot hide withdrawals',()=>{ + for(const phrase of ['requires approval','is conditional on approval','is contingent on acceptance']) for(const target of ['finding','option']) expect(first(change(q=>{if(target==='finding')q.question+='\nThis finding '+phrase+'.';else q.options[0]!.description+='\nThis option '+phrase+'.';}))).toBe(false); + for(const status of ['withdrawn','no longer current']) for(const [open,close]of [['',''],['"','"'],["'","'"],['“','”'],['‘','’']]) expect(first(change(q=>{q.options[0]!.description=q.options[0]!.description!.replace(/\.$/,'')+` This option is ${open}${status}${close}.`;}))).toBe(false); + expect(first(change(q=>{q.options[2]!.description+='\nOnly SessionMint writes.';}))).toBe(false); + expect(first(change(q=>{q.options[0]!.description+='\nDo not inject the cache.';}))).toBe(false); +}); + +test('the injection-only alternative must retain its stated unresolved race',()=>{ + for(const text of ['AuthCache is now serialized.','Only SessionMint writes.']) expect(first(change(q=>{q.options[1]!.description+='\n'+text;}))).toBe(false); + for(const text of ['"AuthCache is now serialized."',"'Only SessionMint writes.'",'ArchiveCache is now serialized.']) expect(first(change(q=>{q.options[1]!.description+='\n'+text;}))).toBe(true); +}); +}); + +describe('eng-architecture-cache-av', () => { +const fixture = fixture_eng_architecture_cache_av; +const fresh=()=>structuredClone(fixture.call) as NativePlanQuestionCall; +const fp=(c:NativePlanQuestionCall)=>nativePlanCallFingerprint(c,0,true); +const accepted=(c:NativePlanQuestionCall)=>engFirstReviewAUQ(fp(c)); +type Q=NativePlanQuestionCall['questions'][number]; +function edit(change:(q:Q,c:NativePlanQuestionCall)=>void){const c=fresh(),q=c.questions[0]!;change(q,c);c.answers={[q.question]:q.options[0]!.label};return c;} + +describe('declarative architecture issue owns the current cache mutation decision',()=>{ + test('the exact completed public decision establishes review before the counter records it',()=>{ + const c=fresh(),before=JSON.stringify(c); + expect(c.toolUseId).toBe('toolu_0147MKgbsvnFruWMDXzQUGVv'); + expect(c.answeredAt).toBe('2026-09-10T23:03:42.025Z'); + expect(accepted(c)).toBe(true); + expect(engSetupAUQ(fp(c))).toBe(false); + expect(planCountQuestionPhase(fp(c),false,engStep0Boundary,engFirstReviewAUQ,engSetupAUQ)).toEqual({preReview:false,reviewStarted:true}); + expect(JSON.stringify(c)).toBe(before); + }); + test('actor and cache renaming, citation changes, decision ordinals and offered deferral keep meaning',()=>{ + const rename=JSON.parse(JSON.stringify(fresh()).replaceAll('AuthBroker','CredentialReader').replaceAll('SessionMint','SessionWriter').replaceAll('AuthCache','TenantCache')); + expect(accepted(rename)).toBe(true); + expect(accepted(edit(q=>{q.question=q.question.replaceAll('PLAN.md:19-20','docs/REVISED.md:31-33').replaceAll('PLAN.md:10','docs/REVISED.md:12');}))).toBe(true); + expect(accepted(edit(q=>{q.header='Arch 7';q.question=q.question.replace('D3 — Architecture issue 1','D22 — Architecture issue 7').replace(/\b1([ABC])\b/g,'7$1');q.options.forEach(o=>{o.label=o.label.replace(/^1/,'7');});}))).toBe(true); + expect(accepted(edit(q=>{q.question=q.question.replace('unserialized mutations\n','unserialized mutations.\n');}))).toBe(true); + expect(accepted(edit(q=>q.options.reverse()))).toBe(true); + for(const option of fresh().questions[0]!.options){const c=fresh();c.answers={[c.questions[0]!.question]:option.label};expect(accepted(c)).toBe(true);} + }); + test('the common native completion and identity gates remain necessary',()=>{ + for(const mutation of [ + (c:NativePlanQuestionCall)=>{c.answered=false;},(c:NativePlanQuestionCall)=>{c.failed=true;}, + (c:NativePlanQuestionCall)=>{delete c.answeredAt;},(c:NativePlanQuestionCall)=>{c.answeredAt='invalid';}, + (c:NativePlanQuestionCall)=>{c.unansweredQuestionIndices=[0];},(c:NativePlanQuestionCall)=>{c.answers={};}, + (c:NativePlanQuestionCall)=>{c.answers={[c.questions[0]!.question]:'unoffered'};}, + (c:NativePlanQuestionCall)=>{c.questions[0]!.multiSelect=true;}, + (c:NativePlanQuestionCall)=>{c.questions.push(structuredClone(c.questions[0]!));}, + ]){const c=fresh();mutation(c);expect(accepted(c)).toBe(false);} + for(const mutation of [ + (f:ReturnType)=>{f.signature='foreign:request';},(f:ReturnType)=>{f.nativeCall!.sessionId='foreign';}, + (f:ReturnType)=>{f.nativeCall!.toolUseId='foreign';},(f:ReturnType)=>{f.nativeQuestionIndex=1;}, + (f:ReturnType)=>{f.options.reverse();}, + ]){const f=fp(fresh());mutation(f);expect(engFirstReviewAUQ(f)).toBe(false);} + }); + test('finding metadata and the current shared-cache premise must agree',()=>{ + for(const mutation of [ + (q:Q)=>{q.header='Arch 2';},(q:Q)=>{q.header='Scope';}, + (q:Q)=>{q.options[0]!.label=q.options[0]!.label.replace(/^1A/,'2A');}, + (q:Q)=>{q.question=q.question.replace('global mutable AuthCache','global mutable OtherCache');}, + (q:Q)=>{q.question=q.question.replace('has AuthBroker and SessionMint','has AuthBroker and AuthBroker');}, + (q:Q)=>{q.question=q.question.replace('nothing serializes','the queue serializes');}, + (q:Q)=>{q.question=q.question.replace('ELI10: PLAN.md','ELI10: If approved, PLAN.md');}, + (q:Q)=>{q.question=q.question.replace(/^ELI10: (.+)$/m,'ELI10: "$1"');}, + (q:Q)=>{q.question=q.question.replace(/^ELI10: (.+)$/m,'> ELI10: $1');}, + (q:Q)=>{q.question='Source example:\n'+q.question;}, + (q:Q)=>{q.question='```text\n'+q.question+'\n```';}, + (q:Q)=>{q.question+='\nELI10: No current race remains.';}, + (q:Q)=>{q.question=q.question.replace('Picture SessionMint','Picture OtherWriter');}, + (q:Q)=>{q.question=q.question.replace('refreshed token for tenant A','refreshed token for tenant B');}, + ])expect(accepted(edit(mutation))).toBe(false); + }); + test('the same offered remedy must inject the named cache, serialize its writes and require tenant identity',()=>{ + for(const mutation of [ + (q:Q)=>{q.options[0]!.label=q.options[0]!.label.replace('Inject AuthCache','Inject OtherCache');}, + (q:Q)=>{q.options[0]!.label=q.options[0]!.label.replace('by constructor','through a global export');}, + (q:Q)=>{q.options[0]!.label=q.options[0]!.label.replace('owns all writes','accepts unowned writes');}, + (q:Q)=>{q.options[0]!.label=q.options[0]!.label.replace('serializes per tenant key','leaves writes unordered');}, + (q:Q)=>{q.options[0]!.label=q.options[0]!.label.replace('requires tenant context','allows missing tenant context');}, + (q:Q)=>{q.options[0]!.description=q.options[0]!.description!.replace('run in order through one owner','run concurrently through both services');}, + (q:Q)=>{q.options[0]!.description=q.options[0]!.description!.replace('fresh AuthCache per case','shared AuthCache for all cases');}, + (q:Q)=>{q.options[0]!.description=q.options[0]!.description!.replace('No method accepts a call without','Every method accepts a call without');}, + (q:Q)=>{q.options[0]!.description='Historical example: '+q.options[0]!.description;}, + (q:Q)=>{q.options[0]!.description='If approved: '+q.options[0]!.description;}, + (q:Q)=>{q.options[0]!.description='> '+q.options[0]!.description;}, + (q:Q)=>{q.options[0]!.description+='\nAuthBroker still writes directly.';}, + (q:Q)=>{q.options[0]!.description+='\nSerialization is optional.';}, + ])expect(accepted(edit(mutation))).toBe(false); + }); + test('the opposed choice must actually leave the current race open',()=>{ + for(const mutation of [ + (q:Q)=>{q.options[2]!.label='1C: Resolve the race';}, + (q:Q)=>{q.options[2]!.description='The cache is already safe and serialized.';}, + (q:Q)=>{q.options[2]!.description='Historical example: '+q.options[2]!.description;}, + (q:Q)=>{q.options[2]!.description+='\nAuthCache is already serialized.';}, + (q:Q)=>{q.options[2]!.description+='\nOnly AuthBroker writes.';}, + (q:Q)=>{q.options[1]!.description+='\nOnly AuthBroker writes.';}, + (q:Q)=>{q.question+='\nAuthCache now serializes all writes.';}, + (q:Q)=>{q.question+='\nOnly SessionMint writes.';}, + (q:Q)=>{q.question+='\nDo not inject this cache.';}, + ])expect(accepted(edit(mutation))).toBe(false); + }); + test('owned current statuses and approvals override the earlier finding across scalar quote forms',()=>{ + for(const target of [-1,0,1,2])for(const owner of ['This finding','D3','Architecture issue 1'])for(const suffix of [" is 'withdrawn'.",' is “no longer current”.',' is `unproven`.',' is optional.',' requires approval.']){ + const c=edit(q=>{const text='\nAssessment complete; '+owner+suffix;if(target<0)q.question+=text;else q.options[target]!.description+=text;}); + expect(accepted(c)).toBe(false); + } + for(const target of [-1,0,2])for(const text of ['\nPrior note: "This finding is withdrawn."','\n> This finding is withdrawn.','\nA previous reviewer said `This finding is withdrawn.`','\nOtherCache is already serialized.']){ + expect(accepted(edit(q=>{if(target<0)q.question+=text;else q.options[target]!.description+=text;}))).toBe(true); + } + }); +}); +}); + +describe('eng-binding-retry-z', () => { +const captured = captured_eng_binding_retry_z; +const fresh = () => structuredClone(captured[1]!) as NativePlanQuestionCall; +const fp = (c: NativePlanQuestionCall) => nativePlanCallFingerprint(c, 0, true); +const first = (c: NativePlanQuestionCall) => engFirstReviewAUQ(fp(c)); +function question(c: NativePlanQuestionCall, transform: (s: string) => string) { + const q = c.questions[0]!; const answer = c.answers![q.question]!; + q.question = transform(q.question); c.answers = {[q.question]: answer}; return c; +} + +describe('Z Eng shared mutable cache starts substantive review', () => { + test('the actual shared mutable cache risk starts review without an issue label', () => { + expect(engSetupAUQ(fp(fresh()))).toBe(false); + expect(first(fresh())).toBe(true); + expect(planCountQuestionPhase(fp(fresh()), false, engStep0Boundary, engFirstReviewAUQ, engSetupAUQ)) + .toEqual({preReview: false, reviewStarted: true}); + }); + + test('the exact six native calls preserve one setup and all five review obligations', () => { + let started = false; + const phases = captured.map(c => { + const p = planCountQuestionPhase(fp(structuredClone(c) as NativePlanQuestionCall), started, engStep0Boundary, engFirstReviewAUQ, engSetupAUQ); + started = p.reviewStarted; return p.preReview; + }); + expect(phases).toEqual([true, false, false, false, false, false]); + expect(first(structuredClone(captured[0]!) as NativePlanQuestionCall)).toBe(false); + expect(captured[5]!.questions[0]!.header).toBe('TODO: E2E test'); + }); + + test('either offered choice, reordering and a different component retain issue identity', () => { + const c = fresh(); c.questions[0]!.options.reverse(); + for (const option of c.questions[0]!.options) { + c.answers = {[c.questions[0]!.question]: option.label}; expect(first(c)).toBe(true); + } + const varied = question(fresh(), s => s.replace('AuthCache', 'SessionCache').replace('D2', 'D7')); + for (const option of varied.questions[0]!.options) option.description = option.description.replaceAll('AuthCache', 'SessionCache'); + expect(first(varied)).toBe(true); + }); + + test('requires a complete native call and exact offered answer and fingerprint', () => { + for (const mutate of [ + (c: NativePlanQuestionCall) => { c.answered = false; }, + (c: NativePlanQuestionCall) => { c.failed = true; }, + (c: NativePlanQuestionCall) => { delete c.failed; }, + (c: NativePlanQuestionCall) => { delete c.unansweredQuestionIndices; }, + (c: NativePlanQuestionCall) => { c.unansweredQuestionIndices = [0]; }, + (c: NativePlanQuestionCall) => { c.questions[0]!.multiSelect = true; }, + (c: NativePlanQuestionCall) => { c.questions.push(structuredClone(c.questions[0]!)); }, + (c: NativePlanQuestionCall) => { c.questions[0]!.options.push(structuredClone(c.questions[0]!.options[0]!)); }, + (c: NativePlanQuestionCall) => { c.questions[0]!.options[1]!.label = c.questions[0]!.options[0]!.label; }, + (c: NativePlanQuestionCall) => { c.answers = {}; }, + (c: NativePlanQuestionCall) => { c.answers = {[c.questions[0]!.question]: 'Foreign answer'}; }, + ]) { const c = fresh(); mutate(c); expect(first(c)).toBe(false); } + expect(engFirstReviewAUQ({...fp(fresh()), signature: 'foreign:call'})).toBe(false); + expect(engFirstReviewAUQ({...fp(fresh()), nativeCall: undefined})).toBe(false); + expect(engFirstReviewAUQ({...fp(fresh()), options: []})).toBe(false); + const mismatch = fp(fresh()); mismatch.options[0]!.label = 'foreign'; expect(engFirstReviewAUQ(mismatch)).toBe(false); + const wrongIndex = fp(fresh()); wrongIndex.options[0]!.index = 2; expect(engFirstReviewAUQ(wrongIndex)).toBe(false); + }); + + test('requires an affirmative direct risk, not setup, denial, qualification or quotations', () => { + for (const header of ['Scope', 'Approach', 'Next review', 'Onboarding']) { + const c = fresh(); c.questions[0]!.header = header; expect(first(c)).toBe(false); + } + for (const transform of [ + (s: string) => s.replace('Architecture:', 'Approach:'), + (s: string) => s.replace('Two services share', 'If two services share'), + (s: string) => s.replace('Two services share', 'Two services do not share'), + (s: string) => s.replace('can corrupt tenant isolation', 'cannot corrupt tenant isolation'), + (s: string) => s.replace('can corrupt tenant isolation', 'never corrupt tenant isolation'), + (s: string) => s.replace('This is the #1 reliability risk', 'This is not the #1 reliability risk'), + (s: string) => s.replace('plan-eng-shared-mutable-cache', 'plan-eng-setup'), + (s: string) => s.replace('plan-eng-shared-mutable-cache', 'foreign-shared-mutable-cache'), + (s: string) => s.replace(/ ]+>/, ''), + (s: string) => s + ' ', + (s: string) => s + ' Run the next review too.', + (s: string) => '> ' + s, + (s: string) => '```text\n' + s + '\n```', + ]) expect(first(question(fresh(), transform))).toBe(false); + }); + + test('the complete offered remedies stay tied to the same dependency and affirmative risk', () => { + for (const [index, transform] of [ + [0, (s: string) => s.replace('The plan is updated', 'The plan is not updated')], + [0, (s: string) => s.replace('pass AuthCache', 'pass DifferentCache')], + [0, (s: string) => s.replace('No module-level mutable export.', 'Keep the module-level mutable export.')], + [1, (s: string) => s.replace('still couples both services', 'does not couple both services')], + [2, (s: string) => s.replace('as a known risk', 'as a dismissed risk')], + [0, (s: string) => s + ' Also grant every tenant access.'], + [1, (s: string) => s + ' Also approve the missing timeout policy.'], + [0, (s: string) => '> ' + s], + [2, (s: string) => '```text\n' + s + '\n```'], + ] as const) { + const c = fresh(); const option = c.questions[0]!.options[index]!; + option.description = transform(option.description ?? ''); expect(first(c)).toBe(false); + } + const c = fresh(); c.questions[0]!.options[0]!.label = 'Run /office-hours'; + c.answers = {[c.questions[0]!.question]: 'Run /office-hours'}; expect(first(c)).toBe(false); + }); +}); +}); + +describe('eng-binding-z', () => { +const captured = captured_eng_binding_z; +const fresh = () => structuredClone(captured[1]!) as NativePlanQuestionCall; +const fp = (c: NativePlanQuestionCall) => nativePlanCallFingerprint(c, 0, true); +const first = (c: NativePlanQuestionCall) => engFirstReviewAUQ(fp(c)); +function question(c: NativePlanQuestionCall, transform: (s: string) => string) { + const q = c.questions[0]!; const answer = c.answers![q.question]!; + q.question = transform(q.question); c.answers = {[q.question]: answer}; return c; +} + +describe('Z Eng dependency binding starts substantive review', () => { + test('the actual cache dependency decision starts review without an issue label', () => { + expect(engSetupAUQ(fp(fresh()))).toBe(false); + expect(first(fresh())).toBe(true); + expect(planCountQuestionPhase(fp(fresh()), false, engStep0Boundary, engFirstReviewAUQ, engSetupAUQ)) + .toEqual({preReview: false, reviewStarted: true}); + }); + + test('the exact six native calls preserve one setup and all five review obligations', () => { + let started = false; + const phases = captured.map(c => { + const p = planCountQuestionPhase(fp(structuredClone(c) as NativePlanQuestionCall), started, engStep0Boundary, engFirstReviewAUQ, engSetupAUQ); + started = p.reviewStarted; return p.preReview; + }); + expect(phases).toEqual([true, false, false, false, false, false]); + expect(first(structuredClone(captured[0]!) as NativePlanQuestionCall)).toBe(false); + expect(captured[5]!.questions[0]!.header).toBe('TODO: Timeout'); + }); + + test('either offered choice, reordering and a different component retain issue identity', () => { + const c = fresh(); c.questions[0]!.options.reverse(); + for (const option of c.questions[0]!.options) { + c.answers = {[c.questions[0]!.question]: option.label}; expect(first(c)).toBe(true); + } + const varied = question(fresh(), s => s.replace('AuthBroker', 'SessionGateway').replace('D2', 'D7')); + for (const option of varied.questions[0]!.options) option.description = option.description.replaceAll('AuthBroker', 'SessionGateway'); + expect(first(varied)).toBe(true); + }); + + test('requires a complete native call and exact offered answer and fingerprint', () => { + for (const mutate of [ + (c: NativePlanQuestionCall) => { c.answered = false; }, + (c: NativePlanQuestionCall) => { c.failed = true; }, + (c: NativePlanQuestionCall) => { delete c.failed; }, + (c: NativePlanQuestionCall) => { delete c.unansweredQuestionIndices; }, + (c: NativePlanQuestionCall) => { c.unansweredQuestionIndices = [0]; }, + (c: NativePlanQuestionCall) => { c.questions[0]!.multiSelect = true; }, + (c: NativePlanQuestionCall) => { c.questions.push(structuredClone(c.questions[0]!)); }, + (c: NativePlanQuestionCall) => { c.questions[0]!.options.push(structuredClone(c.questions[0]!.options[0]!)); }, + (c: NativePlanQuestionCall) => { c.questions[0]!.options[1]!.label = c.questions[0]!.options[0]!.label; }, + (c: NativePlanQuestionCall) => { c.answers = {}; }, + (c: NativePlanQuestionCall) => { c.answers = {[c.questions[0]!.question]: 'Foreign answer'}; }, + ]) { const c = fresh(); mutate(c); expect(first(c)).toBe(false); } + expect(engFirstReviewAUQ({...fp(fresh()), signature: 'foreign:call'})).toBe(false); + expect(engFirstReviewAUQ({...fp(fresh()), nativeCall: undefined})).toBe(false); + expect(engFirstReviewAUQ({...fp(fresh()), options: []})).toBe(false); + const mismatch = fp(fresh()); mismatch.options[0]!.label = 'foreign'; expect(engFirstReviewAUQ(mismatch)).toBe(false); + const wrongIndex = fp(fresh()); wrongIndex.options[0]!.index = 2; expect(engFirstReviewAUQ(wrongIndex)).toBe(false); + }); + + test('setup, foreign, hypothetical, quoted and mixed questions cannot open review', () => { + for (const header of ['Scope', 'Approach', 'Next review', 'Onboarding']) { + const c = fresh(); c.questions[0]!.header = header; expect(first(c)).toBe(false); + } + for (const transform of [ + (s: string) => s.replace('Architecture:', 'Approach:'), + (s: string) => s.replace('How should AuthBroker', 'If needed, how should AuthBroker'), + (s: string) => s.replace('AuthBroker access', 'the whole plan access'), + (s: string) => s.replace('plan-eng-cache-binding', 'plan-eng-setup'), + (s: string) => s.replace('plan-eng-cache-binding', 'foreign-cache-binding'), + (s: string) => s.replace(/ ]+>/, ''), + (s: string) => s + ' ', + (s: string) => s + ' Approve the release too.', + (s: string) => '> ' + s, + (s: string) => '```text\n' + s + '\n```', + ]) expect(first(question(fresh(), transform))).toBe(false); + }); + + test('both descriptions must affirm the existing dependency and remedy without extra obligations', () => { + for (const [index, transform] of [ + [0, (s: string) => s.replace('Eliminates module-level mutable state entirely.', 'Does not eliminate module-level mutable state.')], + [0, (s: string) => s.replace('Eliminates module-level mutable state entirely.', 'If shared state exists, eliminates it.')], + [0, (s: string) => s.replace('AuthBroker receives', 'DifferentComponent receives')], + [1, (s: string) => s.replace('same pattern as the current plan', 'unlike the current plan')], + [1, (s: string) => s.replace('makes tests require module-level mocking', 'does not make tests require module-level mocking')], + [1, (s: string) => s.replace('AuthBroker imports', 'DifferentComponent imports')], + [0, (s: string) => s + ' Also grant every tenant access.'], + [1, (s: string) => s + ' Also approve the missing timeout policy.'], + [0, (s: string) => '> ' + s], + [1, (s: string) => '```text\n' + s + '\n```'], + ] as const) { + const c = fresh(); const option = c.questions[0]!.options[index]!; + option.description = transform(option.description ?? ''); expect(first(c)).toBe(false); + } + const c = fresh(); c.questions[0]!.options[0]!.label = 'Run /office-hours'; + c.answers = {[c.questions[0]!.question]: 'Run /office-hours'}; expect(first(c)).toBe(false); + }); +}); +}); + +describe('eng-cache-brief-am', () => { +const fixture = fixture_eng_cache_brief_am; +const calls=fixture.calls as FP[]; +function edit(change:(q:any,f:FP)=>void):FP { const f=structuredClone(calls[1]!),c=f.nativeCall!,q=c.questions[0]!,selected=q.options.findIndex(o=>o.label===c.answers?.[q.question]);change(q,f);c.answers={[q.question]:q.options[selected]!.label};return nativePlanCallFingerprint(c,f.observedAtMs,f.preReview); } +test('the completed current cache ownership brief starts the engineering review',()=>expect(engFirstReviewAUQ(calls[1]!)).toBe(true)); +test('the earlier whole-plan scope choice does not become a finding',()=>expect(engFirstReviewAUQ(calls[0]!)).toBe(false)); +test('equivalent decision ordinal and current wording retain the owned finding',()=>{ + expect(engFirstReviewAUQ(edit(q=>{q.question=q.question.replace(/^D2/,'D17');q.header='D17 DI';}))).toBe(true); + expect(engFirstReviewAUQ(edit(q=>{q.question=q.question.replace('Right now both services grab','Today both services import').replace('and both write to it.','and both mutate it.');}))).toBe(true); +}); +test('unrelated current decisions remain outside this dependency branch',()=>{ + expect(engFirstReviewAUQ(edit(q=>{q.header='D3 DI';}))).toBe(false); + expect(engFirstReviewAUQ(edit(q=>{q.question=q.question.replace('Architecture finding A1','Architecture finding A2');}))).toBe(false); +}); +const negative:Array<[string,(q:any,f:FP)=>void]>=[ + ['source preface',q=>q.question=q.question.replace('\nELI10:','\nSource excerpt:\nELI10:')], + ['historical current clause',q=>q.question=q.question.replace('ELI10: Right now','ELI10: Previously')], + ['hypothetical current clause',q=>q.question=q.question.replace('ELI10: Right now','ELI10: If approved, right now')], + ['negated current writes',q=>q.question=q.question.replace('and both write to it.','and neither writes to it.')], + ['quoted current assessment',q=>q.question=q.question.replace('ELI10: Right now','ELI10: "Right now').replace('it. Nobody','it." Nobody')], + ['withdrawn finding',q=>q.question+='\nCorrection: this finding is withdrawn.'], + ['resolved current finding',q=>q.question+='\nNo current gap remains.'], + ['foreign cache title',q=>q.question=q.question.replace('Module-level AuthCache','Module-level OtherCache')], + ['source remedy preface',q=>q.options[0].description='Source excerpt:\n'+q.options[0].description], + ['conditional writer ownership',q=>q.options[0].description=q.options[0].description.replace('✅ SessionMint','✅ If SessionMint')], + ['quoted writer ownership',q=>q.options[0].description=q.options[0].description.replace('✅ SessionMint','✅ "SessionMint').replace('not convention.','not convention."')], + ['read-only claim only in con',q=>q.options[0].description=q.options[0].description.replace('✅ SessionMint','❌ SessionMint')], + ['same writable and read-only actor',q=>q.options[0].description=q.options[0].description.replace('AuthBroker gets','SessionMint gets')], + ['withdrawn remedy',q=>q.options[0].description+=' This remedy is withdrawn.'], + ['foreign opposing finding',q=>q.options[2].description=q.options[2].description.replace('Both A1','Both A9')], + ['opposed gap resolved',q=>q.options[2].description+=' No current gap remains.'], + ['conditional opposed gap',q=>q.options[2].description=q.options[2].description.replace('❌ Both A1','❌ If Both A1')], + ['unoffered recommendation',q=>q.question=q.question.replace('Recommendation: A','Recommendation: D')], + ['multiple recommendations',q=>q.options[1].label+=' (recommended)'], + ['unlettered choice',q=>q.options[1].label=q.options[1].label.slice(3)], + ['unanswered',(_,f)=>f.nativeCall!.answered=false], + ['failed',(_,f)=>f.nativeCall!.failed=true], + ['incomplete member',(_,f)=>f.nativeCall!.unansweredQuestionIndices=[0]], + ['invalid completion time',(_,f)=>f.nativeCall!.answeredAt='invalid'], +]; +test.each(negative)('%s cannot supply current owned engineering review',(_,change)=>expect(engFirstReviewAUQ(edit(change))).toBe(false)); +test('whole quoted history and consistent identifiers preserve the current decision',()=>{ + expect(engFirstReviewAUQ(edit(q=>q.question+='\nPrior note: "This finding is withdrawn."'))).toBe(true); + expect(engFirstReviewAUQ(edit(q=>{q.question=q.question.replaceAll('AuthCache','SessionCache').replaceAll('A1','A9');q.options.forEach((o:any)=>o.description=o.description.replaceAll('A1','A9'));}))).toBe(true); + expect(engFirstReviewAUQ(edit(q=>{q.question=q.question.replace(/^D2/,'d2');q.header='d2 DI';}))).toBe(true); +}); +test('current-owner withdrawals and conditional metadata cannot lend review evidence',()=>{ + for(const text of ['Correction: this finding is rejected.','Correction: this remedy is cancelled.','Correction: this finding is "withdrawn".','Correction: this explanation is not current.']) expect(engFirstReviewAUQ(edit(q=>q.question+='\n'+text))).toBe(false); + expect(engFirstReviewAUQ(edit(q=>q.question=q.question.replace('Architecture finding A1','If approved, Architecture finding A1')))).toBe(false); +}); +}); + +describe('eng-cache-owner-an', () => { +const fixture = fixture_eng_cache_owner_an; +type Question = NonNullable['questions'][number]; +const original = () => structuredClone(fixture.fingerprint) as Fingerprint; +function edit(change: (question: Question) => void): Fingerprint { + const fp = original(), call = fp.nativeCall!, question = call.questions[0]!; + const selected = question.options.findIndex(option => option.label === call.answers![question.question]); + change(question); + call.answers = { [question.question]: question.options[selected]!.label }; + fp.options = question.options.map((option, index) => ({ index: index + 1, label: option.label })); + return fp; +} + +test('an owned cache-ownership decision starts review with actors named in the current assessment', () => { + expect(engFirstReviewAUQ(original())).toBe(true); + expect(planCountQuestionPhase(original(), false, engStep0Boundary, engFirstReviewAUQ, engSetupAUQ)) + .toMatchObject({ preReview: false, reviewStarted: true }); + expect(fixture.fingerprint.preReview).toBe(true); +}); + +test('review identity survives equivalent headers, actor names and an offered opposing answer', () => { + for (const header of ['Cache owner', 'Cache ownership', 'Shared cache', 'Issue 1', 'Architecture 1']) + expect(engFirstReviewAUQ(edit(question => { question.header = header; }))).toBe(true); + expect(engFirstReviewAUQ(edit(question => { + question.question = question.question.replaceAll('AuthBroker', 'SessionOwner').replaceAll('SessionMint', 'TokenMinter'); + question.options = question.options.map(option => ({ ...option, + description: option.description?.replaceAll('AuthBroker', 'SessionOwner').replaceAll('SessionMint', 'TokenMinter') })); + }))).toBe(true); + for (const option of original().nativeCall!.questions[0]!.options) { + const fp = original(), call = fp.nativeCall!; + call.answers = { [call.questions[0]!.question]: option.label }; + expect(engFirstReviewAUQ(fp)).toBe(true); + } + expect(engFirstReviewAUQ(edit(question => { question.question += '\nHistorical note: "This finding is withdrawn."'; }))).toBe(true); +}); + +const rejected: Array<[string, (question: Question) => void]> = [ + ['setup header', q => { q.header = 'Outside voices'; }], + ['foreign issue identity', q => { q.header = 'Issue 2'; }], + ['historical title', q => { q.question = 'Historical example:\n' + q.question; }], + ['source assessment', q => { q.question = q.question.replace('\nELI10:', '\nSource:\nELI10:'); }], + ['conditional project', q => { q.question = q.question.replace('Project/branch/task: ', 'Project/branch/task: If approved: '); }], + ['quoted current premise', q => { q.question = q.question.replace(/ELI10: ([^\n]+)/, 'ELI10: "$1"'); }], + ['conditional current premise', q => { q.question = q.question.replace('ELI10: AuthBroker', 'ELI10: If AuthBroker'); }], + ['only one actual actor', q => { q.question = q.question.replace('AuthBroker and SessionMint', 'AuthBroker and AuthBroker'); }], + ['current writes negated', q => { q.question = q.question.replace('both write into', 'neither writes into'); }], + ['withdrawn finding', q => { q.question += '\nThis finding is withdrawn.'; }], + ['rejected numbered issue', q => { q.question += '\nIssue 1 is rejected.'; }], + ['assessment no longer current', q => { q.question += '\nThis assessment is not current.'; }], + ['foreign writer', q => { q.options[0]!.description = q.options[0]!.description!.replace('Only AuthBroker writes', 'Only OtherService writes'); }], + ['foreign producer', q => { q.options[0]!.description = q.options[0]!.description!.replace('SessionMint returns', 'OtherService returns'); }], + ['same writer and producer', q => { q.options[0]!.description = q.options[0]!.description!.replace('SessionMint returns', 'AuthBroker returns'); }], + ['quoted remedy', q => { q.options[0]!.description = '> ' + q.options[0]!.description; }], + ['conditional remedy', q => { q.options[0]!.description = 'If approved: ' + q.options[0]!.description; }], + ['cancelled remedy', q => { q.options[0]!.description += ' This remedy is cancelled.'; }], + ['explicitly rejected injection', q => { q.options[0]!.description += ' Correction: do not inject the adapter.'; }], + ['no opposed action', q => { q.options[2]!.label = 'C) Run another review'; }], + ['quoted deferral', q => { q.options[2]!.description = '> ' + q.options[2]!.description; }], + ['conditional deferral', q => { q.options[2]!.description = 'If approved: ' + q.options[2]!.description; }], + ['no retained race', q => { q.options[2]!.description = q.options[2]!.description!.replace('Race stays open', 'Race is closed'); }], + ['rejected opposed action', q => { q.options[2]!.description += ' This option is rejected.'; }], + ['withdrawn single-writer requirement', q => { q.options[0]!.description += ' The single-writer requirement is withdrawn.'; }], + ['producer also writes', q => { q.options[0]!.description += ' Correction: SessionMint will also write directly to the cache.'; }], + ['retained race closed', q => { q.options[2]!.description += ' Correction: the race is now closed.'; }], +]; +test.each(rejected)('%s does not establish the first review decision', (_, change) => { + expect(engFirstReviewAUQ(edit(change))).toBe(false); +}); + +test('native ownership, completion, answer alignment and dense menus remain required', () => { + const invalid: Array<(fp: Fingerprint) => void> = [ + fp => { fp.nativeCall!.answered = false; }, fp => { fp.nativeCall!.failed = true; }, + fp => { fp.signature = 'foreign:call'; }, fp => { fp.nativeQuestionIndex = 1; }, + fp => { fp.nativeCall!.unansweredQuestionIndices = [0]; }, + fp => { delete fp.nativeCall!.answeredAt; }, fp => { fp.nativeCall!.answers = {}; }, + fp => { fp.options.reverse(); }, + ]; + for (const change of invalid) { + const fp = original(); change(fp); + expect(engFirstReviewAUQ(fp)).toBe(false); + } +}); +}); + +describe('eng-cache-writes-as', () => { +const captured = captured_eng_cache_writes_as; +const actual = () => structuredClone(captured.call) as NativePlanQuestionCall; +function answered(c: NativePlanQuestionCall, index = 0) { + c.answers = { [c.questions[0]!.question]: c.questions[0]!.options[index]!.label }; + return nativePlanCallFingerprint(c, 0, true); +} +function allText(edit: (s: string) => string) { + const c = actual(), q = c.questions[0]!; q.question = edit(q.question); + for (const o of q.options) { o.label = edit(o.label); o.description = edit(o.description ?? ''); } + return c; +} + +test('the exact completed retry starts review with the current cache ownership decision', () => { + const c = actual(), before = JSON.stringify(c), fp = nativePlanCallFingerprint(c, 0, true); + expect(engFirstReviewAUQ(fp)).toBe(true); expect(engSetupAUQ(fp)).toBe(false); + expect(planCountQuestionPhase(fp, false, engStep0Boundary, engFirstReviewAUQ, engSetupAUQ)).toMatchObject({ preReview: false, reviewStarted: true }); + expect(JSON.stringify(c)).toBe(before); expect(captured.provenance.retrospectivePass).toBe(false); +}); + +test('all offered choices and consistently renamed actors retain the same review identity', () => { + for (const reverse of [false, true]) for (let i = 0; i < 3; i++) { + const c = actual(); if (reverse) c.questions[0]!.options.reverse(); + expect(engFirstReviewAUQ(answered(c, i))).toBe(true); + } + for (const [one, two] of [['One', 'Two'], ['$Reader', '_Writer'], ['SessionMint', 'AuthBroker']]) { + const c = allText(t => t.replaceAll('AuthBroker', '__one__').replaceAll('SessionMint', two).replaceAll('__one__', one)); + expect(engFirstReviewAUQ(answered(c))).toBe(true); + } + const c = allText(t => t.replace(/^D2 /, 'D17 ').replace(/\b2([A-C])\b/g, '17$1')); + expect(engFirstReviewAUQ(answered(c))).toBe(true); +}); + +test('native completion, session, exact answer and option binding remain mandatory', () => { + const mutations: Array<(c: NativePlanQuestionCall) => void> = [ + c => { c.answered = false; }, c => { c.failed = true; }, c => { c.answers = {}; }, + c => { c.answers = { [c.questions[0]!.question]: 'unoffered' }; }, c => { c.answers!['foreign'] = 'answer'; }, + c => { c.answeredAt = 'invalid'; }, c => { c.unansweredQuestionIndices = [0]; }, + c => { c.sessionId = ''; }, c => { c.toolUseId = ''; }, c => { c.questions[0]!.multiSelect = true; }, + c => { c.questions.push(structuredClone(c.questions[0]!)); }, c => { c.questions[0]!.options[1]!.label = c.questions[0]!.options[0]!.label; }, + ]; + for (const mutate of mutations) { const c = actual(); mutate(c); expect(engFirstReviewAUQ(nativePlanCallFingerprint(c, 0, true))).toBe(false); } + const fp = answered(actual()); + expect(engFirstReviewAUQ({ ...fp, signature: 'foreign' })).toBe(false); + expect(engFirstReviewAUQ({ ...fp, nativeQuestionIndex: 1 })).toBe(false); + expect(engFirstReviewAUQ({ ...fp, options: [...fp.options].reverse() })).toBe(false); +}); + +test('title, own context, current assessment and two distinct writers are required', () => { + for (const edit of [ + (t: string) => 'Source.\n' + t, (t: string) => '> ' + t, (t: string) => '```\n' + t + '\n```', + (t: string) => t.replace('Who is allowed', 'Who was allowed'), + (t: string) => t.replace('the auth cache?', 'the billing cache?'), + (t: string) => t.replace('Project/branch/task:', 'Earlier review:'), + (t: string) => t.replace('AuthBroker and SessionMint both', 'AuthBroker and AuthBroker both'), + (t: string) => t.replace('both mutating one backing cache', 'both previously mutating one backing cache'), + (t: string) => t.replace('ELI10: Two services', 'ELI10: Source. Two services'), + (t: string) => t.replace('ELI10: Two services', 'ELI10: If approved, two services'), + (t: string) => t.replace('nothing orders their writes.', 'their writes are serialized.'), + (t: string) => t.replace('Project/branch/task: ', 'Project/branch/task: Assuming approval, '), + (t: string) => t.replace('Project/branch/task: ', 'Project/branch/task: Source. '), + (t: string) => t.replace('Multi-tenant Auth Refactor,', 'Multi-tenant Auth Refactor if approved,'), + ]) { const c = actual(); c.questions[0]!.question = edit(c.questions[0]!.question); expect(engFirstReviewAUQ(answered(c))).toBe(false); } + for (const header of ['Scope', 'Issue 1', 'Report', 'Cache examples']) { const c = actual(); c.questions[0]!.header = header; expect(engFirstReviewAUQ(answered(c))).toBe(false); } +}); + +test('owned current status beats a matching assertion while archived and foreign status does not', () => { + for (const status of ['withdrawn', 'superseded', 'rejected', 'cancelled', 'closed', 'hypothetical', 'not current', 'no longer current']) { + for (const [open, close] of [['', ''], ['"', '"'], ["'", "'"], ['“', '”'], ['‘', '’'], ['`', '`']]) { + for (const target of [-1, 0, 2]) for (const owner of ['This finding', 'D2']) { + const c = actual(), q = c.questions[0]!, suffix = `\nCorrection: ${owner} is ${open}${status}${close}.`; + if (target < 0) q.question += suffix; else q.options[target]!.description += suffix; + expect(engFirstReviewAUQ(answered(c)), `${target}: ${owner} ${open}${status}${close}`).toBe(false); + } + } + } + for (const tail of ['D29 is withdrawn.', '> This finding is withdrawn.', 'The prior report said "This finding is withdrawn."', 'An archived review recorded this finding is "withdrawn".', 'An archived review recorded this finding is \'withdrawn\'.', '```\nThis finding is withdrawn.\n```']) { + for (const target of [-1, 0, 2]) { const c = actual(), q = c.questions[0]!; + if (target < 0) q.question += '\n' + tail; else q.options[target]!.description += '\n' + tail; + expect(engFirstReviewAUQ(answered(c)), `${target}: ${tail}`).toBe(true); + } + } +}); + +test('a remedy and opposed choice must bind the same current writers and active race', () => { + for (const index of [0, 2]) for (const prefix of ['Source. ', 'If approved, ', 'Assuming approval, ', 'Historical assessment: ', '> ', '"']) { + const c = actual(), o = c.questions[0]!.options[index]!; o.description = prefix + o.description + (prefix === '"' ? '"' : ''); + expect(engFirstReviewAUQ(answered(c))).toBe(false); + } + for (const index of [0, 1]) for (const name of ['Foreign', 'AuthBroker']) { + const c = actual(), o = c.questions[0]!.options[index]!; o.description = o.description!.replace('SessionMint', name); + expect(engFirstReviewAUQ(answered(c))).toBe(false); + } + for (const [target, tail] of [ + [-1, 'The services no longer mutate the cache.'], [-1, 'Correction: AuthBroker no longer writes to the cache.'], + [-1, 'The writes are now serialized.'], [-1, 'The writers are now serialized.'], [2, 'Correction: The writers are now serialized.'], [0, 'Correction: AuthBroker also writes to the cache.'], + [0, 'The adapter accepts stale writes.'], [0, 'The version check is optional.'], + [2, 'The race is resolved.'], [2, 'Correction: Do not keep both writers.'], + [2, 'Only SessionMint writes to the cache.'], [2, 'Both writers no longer mutate the cache.'], + ] as const) { + const c = actual(), q = c.questions[0]!; if (target < 0) q.question += '\n' + tail; else q.options[target]!.description += '\n' + tail; + expect(engFirstReviewAUQ(answered(c)), `${target}: ${tail}`).toBe(false); + } + for (const i of [0, 2]) { const c = actual(); c.questions[0]!.options[i]!.label = `2${i ? 'C' : 'A'} Record the report`; expect(engFirstReviewAUQ(answered(c))).toBe(false); } +}); +}); + +describe('eng-count-ad-v2', () => { +const captured = captured_eng_count_ad_v2; +const firstCalls = captured.cases.first.calls as NativePlanQuestionCall[]; +const retryCalls = captured.cases.retry.calls as NativePlanQuestionCall[]; +const issue = () => structuredClone(retryCalls[3]!); +const fp = (call: NativePlanQuestionCall) => nativePlanCallFingerprint(call, 0, true); +const isFirst = (call: NativePlanQuestionCall) => engFirstReviewAUQ(fp(call)); +function setupPacket(): NativePlanQuestionCall { + const c = issue(); + c.questions = [ + { header: 'Design doc', question: 'No design doc found for this branch. /office-hours produces sharper review input. Run it first?', + multiSelect: false, options: [{ label: 'Skip — proceed with standard review (recommended)' }, { label: 'Run /office-hours now' }] }, + { header: 'Learnings', question: 'Search learnings from your other projects on this machine?', + multiSelect: false, options: [{ label: 'Enable cross-project learnings (recommended)' }, { label: 'Keep learnings project-scoped only' }] }, + ]; + c.answers = Object.fromEntries(c.questions.map(q => [q.question, q.options[0]!.label])); + return c; +} +function changeQuestion(call: NativePlanQuestionCall, change: (s: string) => string) { + const q = call.questions[0]!, answer = call.answers?.[q.question]; + q.question = change(q.question); call.answers = answer ? { [q.question]: answer } : {}; return call; +} +function census(calls: NativePlanQuestionCall[]) { + let reviewStarted = false; + const counts = { setup: 0, review: 0, administrative: 0 }; + const phases = calls.map(call => { + const phase = planCountQuestionPhase(fp(call), reviewStarted, engStep0Boundary, engFirstReviewAUQ, engSetupAUQ); + reviewStarted = phase.reviewStarted; + counts[phase.administrative ? 'administrative' : phase.preReview ? 'setup' : 'review']++; + return phase; + }); + return { counts, phases }; +} + +describe('Eng AD v2 completed native count evidence', () => { + test('a completed prerequisite and learnings packet closes setup without counting it as a finding', () => { + for (const reverse of [false, true]) { + const c = setupPacket(); if (reverse) c.questions.reverse(); + for (const answer of c.questions.find(q => q.header === 'Learnings')!.options) { + const learning = c.questions.find(q => q.header === 'Learnings')!; + c.answers![learning.question] = answer.label; + const phase = planCountQuestionPhase(fp(c), false, engStep0Boundary, engFirstReviewAUQ, engSetupAUQ); + expect(phase).toEqual({ preReview: true, reviewStarted: true }); + expect(planCountQuestionPhase(fp(issue()), phase.reviewStarted, + engStep0Boundary, engFirstReviewAUQ, engSetupAUQ).preReview).toBe(false); + } + } + }); + + test('partial, ambiguous, foreign and prerequisite-running packets cannot close setup', () => { + for (const mutate of [ + (c: NativePlanQuestionCall) => { c.answered = false; }, + (c: NativePlanQuestionCall) => { c.failed = true; }, + (c: NativePlanQuestionCall) => { c.unansweredQuestionIndices = [1]; }, + (c: NativePlanQuestionCall) => { delete c.answers![c.questions[1]!.question]; }, + (c: NativePlanQuestionCall) => { c.answers![c.questions[1]!.question] = 'unoffered'; }, + (c: NativePlanQuestionCall) => { c.answers![c.questions[0]!.question] = c.questions[0]!.options[1]!.label; }, + (c: NativePlanQuestionCall) => { c.questions[0]!.multiSelect = true; }, + (c: NativePlanQuestionCall) => { c.questions[1]!.options.push({ ...c.questions[1]!.options[0]! }); }, + (c: NativePlanQuestionCall) => { c.questions.push(structuredClone(issue().questions[0]!)); }, + (c: NativePlanQuestionCall) => { c.answeredAt = 'invalid'; }, + ]) { + const c = setupPacket(); mutate(c); expect(engStep0Boundary(fp(c))).toBe(false); + } + expect(engStep0Boundary({ ...fp(setupPacket()), signature: 'foreign:call' })).toBe(false); + expect(engStep0Boundary({ ...fp(setupPacket()), options: [] })).toBe(false); + const c = setupPacket(); + c.questions[1]!.header = 'Issue 1'; + expect(engStep0Boundary(fp(c))).toBe(false); + }); + + test('retry ordinary Issue identity starts review without qids, retaining its later TODO', () => { + const { counts, phases } = census(retryCalls); + expect(counts).toEqual({ setup: 3, review: 6, administrative: 0 }); + expect(phases.slice(3).every(p => !p.preReview && !p.administrative)).toBe(true); + for (const call of retryCalls.slice(3, 8)) expect(isFirst(call)).toBe(true); + expect(retryCalls[8]!.questions[0]!.question).toContain('TODO 1'); + expect(captured.cases.retry.actual.reviewCount).toBe(0); + }); + + test('ordinary issue presentation can vary while completed identity and section number remain bound', () => { + for (const title of ['Issue 1', 'Finding 1.2 (D17)', 'D42 — Issue 1']) { + const call = changeQuestion(issue(), s => s.replace('Issue 1 (D4)', title).replace('AuthCache', 'SessionCache')); + call.questions[0]!.header = title.includes('1.2') ? 'Architecture 1.2' : 'Architecture 1'; + call.questions[0]!.options.reverse(); + for (const option of call.questions[0]!.options) { + call.answers = { [call.questions[0]!.question]: option.label }; + expect(isFirst(call)).toBe(true); + } + } + }); + + test('setup, quoted or mismatched section identities do not start review', () => { + for (const header of ['Scope', 'Approach', 'Next steps', 'Arch 2', 'Example Arch 1', 'TODO 1']) { + const call = issue(); call.questions[0]!.header = header; expect(isFirst(call)).toBe(false); + } + for (const change of [ + (s: string) => '> ' + s, + (s: string) => 'Example: ' + s, + (s: string) => '```text\n' + s + '\n```', + (s: string) => s.replace('Issue 1 (D4)', 'Issue 2 (D4)'), + (s: string) => s + ' ', + ]) expect(isFirst(changeQuestion(issue(), change))).toBe(false); + for (const call of [...firstCalls.slice(0, 4), ...retryCalls.slice(0, 3)]) expect(isFirst(call)).toBe(false); + }); + + test('an Issue heading alone cannot turn a confirmation or report action into a finding', () => { + for (const body of [ + 'No defect remains in the cache. Proceed with the next section?', + 'The cache already serializes writes. Confirm this is accurate?', + 'Should I save the reviewed plan now?', + 'Add a section to the reviewed plan?', + 'Serialize the reviewed plan as JSON for the handoff?', + 'Add the completed tests to this report?', + ]) { + const c = changeQuestion(issue(), () => 'Issue 1 (D4) — ' + body); + c.questions[0]!.options = [ + { label: 'Yes', description: 'Confirm this statement; no new implementation work.' }, + { label: 'No', description: 'Do not confirm; no new implementation work.' }, + ]; + c.answers = { [c.questions[0]!.question]: 'Yes' }; + expect(isFirst(c)).toBe(false); + } + const c = issue(); + c.questions[0]!.options = [{ label: 'Yes', description: 'Confirm; no new work.' }, { label: 'No', description: 'Decline; no new work.' }]; + c.answers = { [c.questions[0]!.question]: 'Yes' }; + expect(isFirst(c)).toBe(false); + }); + + test('new first-finding path requires exact completed native identity and answer', () => { + for (const factory of [issue]) { + const classify = isFirst; + for (const mutate of [ + (c: NativePlanQuestionCall) => { c.answered = false; }, + (c: NativePlanQuestionCall) => { c.failed = true; }, + (c: NativePlanQuestionCall) => { delete c.failed; }, + (c: NativePlanQuestionCall) => { c.sessionId = ''; }, + (c: NativePlanQuestionCall) => { c.toolUseId = ''; }, + (c: NativePlanQuestionCall) => { delete c.answeredAt; }, + (c: NativePlanQuestionCall) => { c.answeredAt = 'invalid'; }, + (c: NativePlanQuestionCall) => { c.unansweredQuestionIndices = [0]; }, + (c: NativePlanQuestionCall) => { delete c.unansweredQuestionIndices; }, + (c: NativePlanQuestionCall) => { c.questions[0]!.multiSelect = true; }, + (c: NativePlanQuestionCall) => { c.questions.push(structuredClone(c.questions[0]!)); }, + (c: NativePlanQuestionCall) => { c.questions[0]!.options[1]!.label = c.questions[0]!.options[0]!.label; }, + (c: NativePlanQuestionCall) => { c.answers = {}; }, + (c: NativePlanQuestionCall) => { c.answers = { [c.questions[0]!.question]: 'Unoffered' }; }, + (c: NativePlanQuestionCall) => { c.answers!.foreign = 'Foreign'; }, + ]) { const c = factory(); mutate(c); expect(classify(c)).toBe(false); } + const classifyFp = engFirstReviewAUQ; + expect(classifyFp({ ...fp(factory()), signature: 'foreign:call' })).toBe(false); + expect(classifyFp({ ...fp(factory()), nativeCall: undefined })).toBe(false); + expect(classifyFp({ ...fp(factory()), nativeQuestionIndex: 1 })).toBe(false); + expect(classifyFp({ ...fp(factory()), options: [] })).toBe(false); + const wrong = fp(factory()); wrong.options[0]!.index = 2; expect(classifyFp(wrong)).toBe(false); + } + }); +}); +}); + +describe('eng-declarative-as', () => { +const captured = captured_eng_declarative_as; +const actual = () => structuredClone(captured.call) as NativePlanQuestionCall; +function answered(c: NativePlanQuestionCall, index = 0) { + c.answers = { [c.questions[0]!.question]: c.questions[0]!.options[index]!.label }; + return nativePlanCallFingerprint(c, 0, true); +} +function edit(replace: (text: string) => string) { + const c = actual(), q = c.questions[0]!; + q.question = replace(q.question); + q.options.forEach(o => { o.label = replace(o.label); o.description = replace(o.description ?? ''); }); + return c; +} + +test('a completed declarative cache issue starts review without a question mark or qid', () => { + const fp = answered(actual()); + expect(engFirstReviewAUQ(fp)).toBe(true); + expect(engSetupAUQ(fp)).toBe(false); + expect(planCountQuestionPhase(fp, false, engStep0Boundary, engFirstReviewAUQ, engSetupAUQ)).toMatchObject({ preReview: false, reviewStarted: true }); + expect(captured.provenance.retrospectivePass).toBe(false); +}); + +test('all answers, option orders, identifiers and consistent ordinals qualify', () => { + for (const reverse of [false, true]) for (let i = 0; i < 3; i++) { + const c = actual(); if (reverse) c.questions[0]!.options.reverse(); + expect(engFirstReviewAUQ(answered(c, i))).toBe(true); + } + for (const name of ['TenantCache', '$Shared', '_Store']) expect(engFirstReviewAUQ(answered(edit(t => t.replaceAll('AuthCache', name))))).toBe(true); + for (const [one, two] of [['First', 'Second'], ['Z_store', '$Reader'], ['SessionMint', 'AuthBroker']]) { + expect(engFirstReviewAUQ(answered(edit(t => t.replaceAll('AuthBroker', '__first__').replaceAll('SessionMint', two).replaceAll('__first__', one))))).toBe(true); + } + for (const kind of ['Issue', 'Finding']) expect(engFirstReviewAUQ(answered(edit(t => t.replace('Issue 1 ', `${kind} 17 `).replace(/\b1([A-C])\b/g, '17$1'))))).toBe(true); +}); + +test('native completion and matching answered menu remain required', () => { + const mutations: Array<(c: NativePlanQuestionCall) => void> = [ + c => { c.answered = false; }, c => { c.failed = true; }, c => { c.answers = {}; }, + c => { c.answers = { [c.questions[0]!.question]: 'unoffered' }; }, c => { c.answers!['foreign'] = 'answer'; }, + c => { c.answeredAt = 'invalid'; }, c => { c.unansweredQuestionIndices = [0]; }, + c => { c.sessionId = ''; }, c => { c.toolUseId = ''; }, c => { c.questions[0]!.multiSelect = true; }, + c => { c.questions.push(structuredClone(c.questions[0]!)); }, + c => { c.questions[0]!.options[1]!.label = c.questions[0]!.options[0]!.label; }, + ]; + for (const mutate of mutations) { const c = actual(); mutate(c); expect(engFirstReviewAUQ(nativePlanCallFingerprint(c, 0, true))).toBe(false); } + const fp = answered(actual()); + expect(engFirstReviewAUQ({ ...fp, signature: 'foreign' })).toBe(false); + expect(engFirstReviewAUQ({ ...fp, nativeQuestionIndex: 1 })).toBe(false); + expect(engFirstReviewAUQ({ ...fp, options: [...fp.options].reverse() })).toBe(false); +}); + +test('the brief must own a current architecture assessment and distinct writers', () => { + const changes = [ + (t: string) => 'Example: ' + t, (t: string) => '> ' + t, (t: string) => '```\n' + t + '\n```', + (t: string) => t.replace('Section 1 Architecture', 'Section 1 Administration'), + (t: string) => t.replace('ELI10: Two services', 'ELI10: If two services'), + (t: string) => t.replace('ELI10: Two services', 'ELI10: Source example: Two services'), + (t: string) => t.replace('AuthBroker and SessionMint share', 'AuthBroker and AuthBroker share'), + (t: string) => t.replace('share a global mutable', 'used to share a global mutable'), + (t: string) => t.replace('writes can interleave', 'writes are serialized').replace('no ordering', 'per-key ordering'), + (t: string) => t.replace('Project/branch/task:', 'Historical assessment:'), + ]; + for (const change of changes) { const c = actual(); c.questions[0]!.question = change(c.questions[0]!.question); expect(engFirstReviewAUQ(answered(c))).toBe(false); } + for (const header of ['Setup', 'TODOs', 'Issue 2', 'Review report']) { const c = actual(); c.questions[0]!.header = header; expect(engFirstReviewAUQ(answered(c))).toBe(false); } +}); + +test('same-decision withdrawals and contrary current state invalidate the issue', () => { + for (const status of ['withdrawn', 'superseded', 'rejected', 'cancelled', 'resolved', 'closed', 'not current', 'no longer current']) { + for (const literal of [status, `"${status}"`, `“${status}”`, `'${status}'`, `‘${status}’`, '`' + status + '`']) for (const target of ['question', 'remedy', 'unchanged']) { + const c = actual(), q = c.questions[0]!, suffix = ` This finding is ${literal}.`; + if (target === 'question') q.question += suffix; else q.options[target === 'remedy' ? 0 : 2]!.description += suffix; + expect(engFirstReviewAUQ(answered(c))).toBe(false); + } + } + for (const contradiction of ['The cache is no longer global.', 'The services no longer mutate shared state.', 'No current risk remains.']) { + const c = actual(); c.questions[0]!.question += '\n' + contradiction; expect(engFirstReviewAUQ(answered(c))).toBe(false); + } +}); + +test('technical options cannot be quoted, hypothetical, mismatched or cancelled', () => { + for (const index of [0, 2]) for (const frame of ['Example: ', 'If approved: ', 'Source excerpt: ', '> ']) { + const c = actual(); c.questions[0]!.options[index]!.description = frame + c.questions[0]!.options[index]!.description; + expect(engFirstReviewAUQ(answered(c))).toBe(false); + } + for (const [index, suffix] of [[0, ' Correction: Do not remove the module-level export.'], [0, ' The module export remains.'], [0, ' Writes remain unordered.'], [2, ' Correction: Do not proceed as written.'], [2, ' The race is resolved.']] as const) { + const c = actual(); c.questions[0]!.options[index]!.description += suffix; expect(engFirstReviewAUQ(answered(c))).toBe(false); + } + for (const index of [0, 2]) { const c = actual(); c.questions[0]!.options[index]!.label = '1' + (index ? 'C' : 'A') + ') Record in report'; expect(engFirstReviewAUQ(answered(c))).toBe(false); } + const c = actual(); c.questions[0]!.options[0]!.description = c.questions[0]!.options[0]!.description!.replaceAll('AuthCache', 'UnrelatedCache'); + expect(engFirstReviewAUQ(answered(c))).toBe(false); +}); +test('current named-owner contradictions are distinct from foreign and archived references', () => { + for (const statement of ['AuthCache is no longer global.', 'AuthCache is no longer mutable.', 'AuthBroker no longer mutates the cache.', 'SessionMint no longer writes to the cache.', 'D4 is withdrawn.', 'Correction: This finding is withdrawn.', 'Correction: Issue 1 is "withdrawn".', 'Correction: AuthCache is no longer global.', "Issue 1 is 'withdrawn'.", 'This finding is “withdrawn”.']) { + for (const boundary of ['\n', '; ']) { + const c = actual(); c.questions[0]!.question += boundary + statement; + expect(engFirstReviewAUQ(answered(c))).toBe(false); + } + } + for (const statement of ['Issue 19 is withdrawn.', 'D42 is withdrawn.', 'AnotherCache is no longer global.', 'An archived review recorded this finding is "withdrawn".', "An archived review recorded this finding is 'withdrawn'.", 'The prior report said "This finding is withdrawn."', '> This finding is withdrawn.']) { + const c = actual(); c.questions[0]!.question += '\n' + statement; + expect(engFirstReviewAUQ(answered(c))).toBe(true); + } + for (const replacement of ['an unrelated billing cache', 'a different cache', 'an OtherCache']) { + const c = actual(); c.questions[0]!.options[2]!.description = c.questions[0]!.options[2]!.description!.replace('an auth cache', replacement); + expect(engFirstReviewAUQ(answered(c))).toBe(false); + } + const c = actual(); c.questions[0]!.options[2]!.description = c.questions[0]!.options[2]!.description!.replace('an auth cache', 'an AuthCache'); + expect(engFirstReviewAUQ(answered(c))).toBe(true); +}); + + +test('the unchanged option cannot contradict its own remaining cache risk', () => { + for (const statement of ['Correction: The writers are now serialized.', 'AuthCache is no longer global.', 'The cache is removed.']) { + const c = actual(); c.questions[0]!.options[2]!.description += '\n' + statement; + expect(engFirstReviewAUQ(answered(c))).toBe(false); + } +}); +}); + +describe('eng-declared-retry-at', () => { +const captured = captured_eng_declared_retry_at; +const first=()=>structuredClone(captured) as NativePlanQuestionCall; +const fp=(c=first())=>nativePlanCallFingerprint(c,0,true); +const classify=(c=first())=>engFirstReviewAUQ(fp(c)); +function mutated(fn:(c:NativePlanQuestionCall)=>void){const c=first();fn(c);return c;} +function text(fn:(s:string)=>string){return mutated(c=>{const q=c.questions[0]!,answer=c.answers![q.question]!;q.question=fn(q.question);c.answers={[q.question]:answer};});} +describe('declarative engineering retry choice',()=>{ + test('recognizes the actual answered finding without requiring a question mark',()=>{ + expect(classify()).toBe(true);expect(engSetupAUQ(fp())).toBe(false); + }); + test('consistent issue numbers, option order and chosen option may vary',()=>{ + const c=first(),q=c.questions[0]!;q.question=q.question.replace('D2 — Issue 1:','D8 — Issue 4:').replace('Recommendation: 1A','Recommendation: 4A');q.header='Issue 4';q.options.forEach(o=>{o.label=o.label.replace(/^1/,'4');});q.options.reverse(); + for(const o of q.options){c.answers={[q.question]:o.label};expect(classify(c)).toBe(true);} + }); + test('native completion, original menu and response ownership remain mandatory',()=>{ + for(const change of [ + (c:NativePlanQuestionCall)=>{c.answered=false;},(c:NativePlanQuestionCall)=>{c.failed=true;},(c:NativePlanQuestionCall)=>{delete c.failed;},(c:NativePlanQuestionCall)=>{c.answeredAt='invalid';},(c:NativePlanQuestionCall)=>{c.sessionId='';},(c:NativePlanQuestionCall)=>{c.toolUseId='';},(c:NativePlanQuestionCall)=>{c.answers={};},(c:NativePlanQuestionCall)=>{c.answers={[c.questions[0]!.question]:'unoffered'};},(c:NativePlanQuestionCall)=>{c.unansweredQuestionIndices=[0];},(c:NativePlanQuestionCall)=>{c.questions.push(structuredClone(c.questions[0]!));},(c:NativePlanQuestionCall)=>{c.questions[0]!.multiSelect=true;}, + ])expect(classify(mutated(change))).toBe(false); + for(const f of [{...fp(),signature:'foreign:call'},{...fp(),nativeQuestionIndex:1},{...fp(),options:[...fp().options].reverse()}])expect(engFirstReviewAUQ(f)).toBe(false); + }); + test('issue identity and report administration cannot substitute for the finding',()=>{ + for(const [a,b] of [['Issue 1:','Issue 0:'],['Issue 1:','Issue 01:'],['Issue 1:','Finding 1:'],['PLAN.md:6-8','PLAN.md:0-8'],['D2 —','D02 —']])expect(classify(text(s=>s.replace(a!,b!)))).toBe(false); + for(const h of ['Issue 2','Routing','Report','Scope'])expect(classify(mutated(c=>{c.questions[0]!.header=h;}))).toBe(false); + expect(classify(text(s=>s.replace(/^Project\/branch\/task:.*\n/m,'')))).toBe(false); + expect(classify(text(s=>s.replace('ELI10:','Project/branch/task: unrelated\nELI10:')))).toBe(false); + expect(classify(mutated(c=>{c.questions[0]!.options[0]!.label='1A) Continue the review';c.answers={[c.questions[0]!.question]:c.questions[0]!.options[0]!.label};}))).toBe(false); + }); + test('quoted, hypothetical and closed current findings remain excluded',()=>{ + for(const prefix of ['Source excerpt: ','Earlier review assessment: ','If approved, ','Provided approval, ']){ + expect(classify(text(s=>s.replace('ELI10: ','ELI10: '+prefix)))).toBe(false); + for(const i of [0,2])expect(classify(mutated(c=>{c.questions[0]!.options[i]!.description=prefix+c.questions[0]!.options[i]!.description;}))).toBe(false); + } + for(const status of ['withdrawn','resolved','"closed"','“superseded”']){ + expect(classify(text(s=>s+` This finding is ${status}.`))).toBe(false); + for(const i of [0,2])expect(classify(mutated(c=>{c.questions[0]!.options[i]!.description+=` This option is ${status}.`;}))).toBe(false); + } + expect(classify(text(s=>s+'\n"Earlier review assessment: This finding is withdrawn."'))).toBe(true); + expect(classify(text(s=>s+'\nCorrection: retry scheduling no longer runs inside each worker.'))).toBe(false); + }); + test('remedy and unchanged choice each retain their own current consequence',()=>{ + for(const [i,a,b] of [[0,'come from the library','might be evaluated later'],[0,'a pure function, trivially unit-tested','five separate implementations'],[2,'a crash or deploy mid-backoff drops the retry','a crash or deploy preserves every retry']] as const)expect(classify(mutated(c=>{const o=c.questions[0]!.options[i]!;o.description=o.description!.replace(a,b);}))).toBe(false); + expect(classify(mutated(c=>{c.questions[0]!.options[0]!.description+='\nCorrection: the library will not own persistence.';}))).toBe(false); + expect(classify(mutated(c=>{c.questions[0]!.options[0]!.description+='\nCorrection: do not use the library retry hook.';}))).toBe(false); + expect(classify(mutated(c=>{c.questions[0]!.options[2]!.description+='\nCorrection: the per-worker scheduler is now crash-safe.';}))).toBe(false); + }); +}); + +describe('current owner status and approval boundaries',()=>{ +const ownedStatusCases:Array<{name:string,expected:boolean,edit:(c:any)=>void}>=[];const add=(name:string,expected:boolean,edit:(c:any)=>void)=>ownedStatusCases.push({name,expected,edit}); +const question=(c:any,suffix:string)=>{const q=c.questions[0],answer=c.answers[q.question];q.question+=suffix;c.answers={[q.question]:answer}}; +add('exact completed declarative choice',true,()=>{}); +for(const owner of ['This finding','Issue 1','D2'])for(const status of ['withdrawn','not current','no longer current'])for(const quote of ['',"'",'‘'])add(`current ${owner} ${quote}${status}`,false,c=>question(c,`\n${owner} is ${quote}${status}${quote==='‘'?'’':quote}.`)); +for(const i of [0,2])for(const status of ['withdrawn','not current','no longer current'])for(const quote of ['',"'",'‘'])add(`option ${i} ${quote}${status}`,false,c=>{c.questions[0].options[i].description+=`\nThis option is ${quote}${status}${quote==='‘'?'’':quote}.`}); +for(const i of [0,2])for(const condition of ['This option applies only if approved.','This option is conditional on approval.','If approved, proceed with this option.'])add(`option ${i} condition ${condition}`,false,c=>{c.questions[0].options[i].description+='\n'+condition}); +for(const condition of ['This finding applies only if approved.','This finding is conditional on approval.'])add('finding condition '+condition,false,c=>question(c,'\n'+condition)); +for(const owner of ['Issue 2','D3'])add('foreign closed owner '+owner,true,c=>question(c,`\n${owner} is withdrawn.`)); +for(const i of [0,2])add(`quoted historical option${i}`,true,c=>{c.questions[0].options[i].description+='\nEarlier review assessment: "This option is withdrawn."';}); +add('quoted historical finding',true,c=>question(c,'\n"Earlier review assessment: This finding is withdrawn."')); +add('quoted title',false,c=>{const q=c.questions[0],a=c.answers[q.question];q.question=q.question.replace(/^(.*)\n/,'"$1"\n');c.answers={[q.question]:a}}); +add('absent completion',false,c=>{c.answered=false});add('failed native result',false,c=>{c.failed=true});add('invalid answer time',false,c=>{c.answeredAt='missing'}); +add('remedy and unchanged outcomes reversed',false,c=>{const o=c.questions[0].options;[o[0].description,o[2].description]=[o[2].description,o[0].description]}); +add('remedy actually declines library persistence',false,c=>{c.questions[0].options[0].description+='\nThe library will not own persistence.'}); +add('unchanged is now crash safe',false,c=>{c.questions[0].options[2].description+='\nThe scheduler is now crash-safe.'}); +for(const control of ownedStatusCases)test(control.name,()=>{const call=first();control.edit(call);expect(classify(call)).toBe(control.expected);}); +}); + + test('bold current owners keep their scalar status before source quotes are removed',()=>{ + for(const owner of ['This finding','D2']){ + expect(classify(text(s=>s+`\n**${owner}** is 'withdrawn'.`))).toBe(false); + expect(classify(text(s=>s+`\n**${owner}** is ‘withdrawn’.`))).toBe(false); + } + for(const i of [0,2])expect(classify(mutated(c=>{c.questions[0]!.options[i]!.description+=`\n**This option** is 'withdrawn'.`;}))).toBe(false); + expect(classify(text(s=>s+'\n"Earlier review assessment: **This finding** is withdrawn."'))).toBe(true); + }); +}); + +describe('eng-first-category-af', () => { +const captured = captured_eng_first_category_af; +const actual = () => structuredClone(captured.fingerprint.nativeCall) as NativePlanQuestionCall; + +test('actual completed Architecture issue starts review', () => { + const fp = nativePlanCallFingerprint(actual(), 0, true); + expect(engFirstReviewAUQ(fp)).toBe(true); + expect(engSetupAUQ(fp)).toBe(false); + expect(planCountQuestionPhase(fp, false, engStep0Boundary, engFirstReviewAUQ, engSetupAUQ)) + .toMatchObject({ preReview: false, reviewStarted: true }); + expect(captured.provenance.retrospectivePass).toBe(false); +}); + +function answer(c: NativePlanQuestionCall, index = 0) { + c.answers = {[c.questions[0]!.question]: c.questions[0]!.options[index]!.label}; + return nativePlanCallFingerprint(c, 0, true); +} + +test('all offered choices and menu orders remain substantive decisions', () => { + for (const reverse of [false, true]) for (let index = 0; index < 3; index++) { + const c = actual(); if (reverse) c.questions[0]!.options.reverse(); + expect(engFirstReviewAUQ(answer(c, index))).toBe(true); + } +}); + +test('identifier spelling, writer order and matching issue numbers are incidental', () => { + for (const [left, right] of [['TenantReader', 'SessionWriter'], ['Z_store', '$AStore'], ['SessionMint', 'AuthBroker']]) { + const c = actual(); const q = c.questions[0]!; + q.question = q.question.replace('AuthBroker and SessionMint', `${left} and ${right}`); + expect(engFirstReviewAUQ(answer(c))).toBe(true); + } + for (const kind of ['Issue', 'Finding']) { + const c = actual(); c.questions[0]!.question = c.questions[0]!.question.replace('D4 — Issue 1', `D87 — ${kind} 12.3`); + c.questions[0]!.header = `${kind} 12.3`; + expect(engFirstReviewAUQ(answer(c))).toBe(true); + } +}); + +test('native completion, timestamp, answer and menu identity remain mandatory', () => { + const mutations: Array<(c: NativePlanQuestionCall) => void> = [ + c => { c.answered = false; }, c => { c.failed = true; }, c => { c.answers = {}; }, + c => { c.answers = {[c.questions[0]!.question]: 'not offered'}; }, + c => { c.questions[0]!.question += ' changed'; }, + c => { c.unansweredQuestionIndices = [0]; }, c => { c.answeredAt = 'invalid'; }, + c => { c.sessionId = ''; }, c => { c.toolUseId = ''; }, + c => { c.questions[0]!.multiSelect = true; }, + c => { c.questions.push(structuredClone(c.questions[0]!)); }, + c => { c.questions[0]!.options[1]!.label = c.questions[0]!.options[0]!.label; }, + ]; + for (const mutate of mutations) { + const c = actual(); mutate(c); expect(engFirstReviewAUQ(nativePlanCallFingerprint(c, 0, true))).toBe(false); + } + const fp = nativePlanCallFingerprint(actual(), 0, true); + expect(engFirstReviewAUQ({...fp,signature:'foreign'})).toBe(false); + expect(engFirstReviewAUQ({...fp,nativeQuestionIndex:1})).toBe(false); + expect(engFirstReviewAUQ({...fp,options:[...fp.options].reverse()})).toBe(false); +}); + +test('administrative, TODO, uncertain and quoted contexts cannot borrow technical labels', () => { + const base = actual().questions[0]!.question.split('\n')[0]!; + const titles = [ + 'D4 — Issue 1 (Architecture): Record the completed review in TODOs?', + 'D4 — Issue 1 (Architecture): Confirm that the shared cache review is complete?', + 'D4 — Issue 1 (Architecture): Which review runs next?', + base.replace('AuthBroker and SessionMint both mutate', 'If AuthBroker and SessionMint both mutate'), + base.replace('AuthBroker and SessionMint', 'AuthBroker and AuthBroker'), + base.replace('with no owner and no serialization', 'with an owner and per-key serialization'), + 'Example: ' + base, '> ' + base, '```\n' + base, + base.replace('How should shared-state access be structured?', 'Should the review report record this finding?'), + ]; + for (const title of titles) { + const c = actual(); c.questions[0]!.question = title; + expect(engFirstReviewAUQ(answer(c))).toBe(false); + } + for (const header of ['Issue 2', 'Issue 1.2', 'TODOs', 'Setup', 'Next review']) { + const c = actual(); c.questions[0]!.header = header; + expect(engFirstReviewAUQ(answer(c))).toBe(false); + } +}); + +test('opposed implementation choices cannot be replaced by report or workflow choices', () => { + for (const labels of [ + ['Record in report', 'Defer the report', 'Keep the report'], + ['Run Eng next', 'Run Design next', 'Keep reviewing manually'], + ]) { + const c = actual(); c.questions[0]!.options.forEach((o, i) => {o.label = labels[i]!;}); + expect(engFirstReviewAUQ(answer(c))).toBe(false); + } + const c = actual(); c.questions[0]!.options[0]!.description = ''; + expect(engFirstReviewAUQ(answer(c))).toBe(false); +}); +}); + +describe('eng-first-review-t', () => { +const calls: NativePlanQuestionCall[] = JSON.parse(readFileSync(join(import.meta.dir, 'fixtures/eng-batching-t-calls.json'), 'utf8')); +const fresh = () => structuredClone(calls[0]!); +const first = (call: NativePlanQuestionCall) => engFirstReviewAUQ(nativePlanCallFingerprint(call, 0, true)); +function question(call: NativePlanQuestionCall, text: string) { + const q = call.questions[0]!; const answer = call.answers![q.question]; + call.answers = { [text]: answer! }; q.question = text; +} + +describe('T Eng first architecture choice', () => { + test('the actual first architecture issue starts review on this call', () => { + const call = fresh(); const fp = nativePlanCallFingerprint(call, 0, true); + expect(engSetupAUQ(fp)).toBe(false); + expect(first(call)).toBe(true); + expect(planCountQuestionPhase(fp, false, engStep0Boundary, engFirstReviewAUQ, engSetupAUQ)) + .toEqual({ preReview: false, reviewStarted: true }); + }); + + test('all five exact native decisions remain separate review calls', () => { + let started = false; + const phases = calls.map(call => { + const fp = nativePlanCallFingerprint(call, 0, !started); + const phase = planCountQuestionPhase(fp, started, engStep0Boundary, engFirstReviewAUQ, engSetupAUQ); + started = phase.reviewStarted; return phase.preReview; + }); + expect(phases).toEqual([false, false, false, false, false]); + }); + + test('offered answer identity survives reorder and either alternative decision', () => { + const call = fresh(); call.questions[0]!.options.reverse(); + expect(first(call)).toBe(true); + for (const option of call.questions[0]!.options) { + call.answers![call.questions[0]!.question] = option.label; + expect(first(call)).toBe(true); + } + }); + + test('requires one completed native question and exact offered answer', () => { + const variants: Array<(c: NativePlanQuestionCall) => void> = [ + c => { c.answered = false; }, c => { c.failed = true; }, + c => { delete c.unansweredQuestionIndices; }, c => { c.unansweredQuestionIndices = [0]; }, + c => { c.answers = {}; }, c => { c.answers![c.questions[0]!.question] = 'Foreign answer'; }, + c => { c.questions.push(structuredClone(c.questions[0]!)); }, + c => { c.questions[0]!.multiSelect = true; }, + c => { c.questions[0]!.options[1]!.label = c.questions[0]!.options[0]!.label; }, + ]; + for (const change of variants) { const call = fresh(); change(call); expect(first(call)).toBe(false); } + const fp = nativePlanCallFingerprint(fresh(), 0, true); fp.signature = 'foreign:identity'; + expect(engFirstReviewAUQ(fp)).toBe(false); + delete fp.nativeCall; expect(engFirstReviewAUQ(fp)).toBe(false); + }); + + test('whole-plan approach, setup, missing identity and quoted examples cannot start review', () => { + for (const header of ['Approach', 'Scope', 'Routing rules', 'Next review']) { + const call = fresh(); call.questions[0]!.header = header; expect(first(call)).toBe(false); + } + const original = fresh().questions[0]!.question; + for (const text of [ + original.replace('arch-retry-scheduler', 'arch-setup'), original.replace(/ ]+>/, ''), + original.replace('Architecture: Custom retry scheduler', 'Approach: Which whole-plan direction'), + '> ' + original, '```text\n' + original + '\n```', original + ' Should we start another review?', + ]) { const call = fresh(); question(call, text); expect(first(call)).toBe(false); } + }); + + test('requires an affirmative existing defect, not a neutral or negated comparison', () => { + for (const description of [ + 'Both implementations are equally valid choices.', + 'Each worker gets its own copy. There is no DRY violation.', + fresh().questions[0]!.options[2]!.description!.replace('acknowledged DRY violation', 'no DRY violation'), + fresh().questions[0]!.options[2]!.description!.replace('Creates 5 divergence points', 'No longer creates 5 divergence points'), + '```text\n' + fresh().questions[0]!.options[2]!.description + '\n```', + ]) { const call = fresh(); call.questions[0]!.options[2]!.description = description; expect(first(call)).toBe(false); } + const call = fresh(); call.questions[0]!.options[0]!.label = 'Run /office-hours'; + call.answers![call.questions[0]!.question] = 'Run /office-hours'; expect(first(call)).toBe(false); + }); +}); +}); + +describe('eng-injected-export-aq', () => { +const fixture = fixture_eng_injected_export_aq; +const calls=()=>structuredClone(fixture.calls) as NativePlanQuestionCall[]; +const first=()=>calls()[1]!; +const fp=(c=first())=>nativePlanCallFingerprint(c,0,true); +const classify=(c=first())=>engFirstReviewAUQ(fp(c)); +function mutate(change:(c:NativePlanQuestionCall)=>void){const c=first();change(c);return c;} +function text(change:(s:string)=>string){return mutate(c=>{const q=c.questions[0]!,answer=c.answers![q.question]!;q.question=change(q.question);c.answers={[q.question]:answer};});} +describe('AQ current injected-export architecture decision',()=>{ + test('exact eight owned calls start review only at D2 and preserve scope first',()=>{ + let started=false;const rows=calls().map(c=>{const p=planCountQuestionPhase(fp(c),started,engStep0Boundary,engFirstReviewAUQ,engSetupAUQ);started=p.reviewStarted;return p;}); + expect(rows.map(r=>r.preReview)).toEqual([true,false,false,false,false,false,false,false]); + expect(classify()).toBe(true);expect(engSetupAUQ(fp())).toBe(false); + expect(calls().map(c=>classify(c))).toEqual([false,true,false,false,false,false,false,false]); + }); + test('consistent named actors, cache, issue and decision numbers may vary',()=>{ + const c=first(),q=c.questions[0]!; + const rename=(s:string)=>s.replaceAll('AuthCache','TokenStore').replaceAll('AuthBroker','LoginReader').replaceAll('SessionMint','SessionWriter').replace('D2 — Issue 1:','D8 — Issue 3:'); + q.question=rename(q.question);q.header='Architecture 3';for(const o of q.options){o.label=rename(o.label).replace(/^1/,'3');o.description=rename(o.description??'');} + q.options.reverse();for(const o of q.options){c.answers={[q.question]:o.label};expect(classify(c)).toBe(true);} + }); + test('the current global premise admits Today and both constructor actor orders',()=>{ + expect(classify(text(s=>s.replace('Right now the cache','Today the cache')))).toBe(true); + expect(classify(mutate(c=>{c.questions[0]!.options[0]!.description=c.questions[0]!.options[0]!.description!.replace('AuthBroker and SessionMint constructors','SessionMint and AuthBroker constructors');}))).toBe(true); + }); + test('requires complete single-question native ownership and an offered answer',()=>{ + for(const change of [(c:NativePlanQuestionCall)=>{c.answered=false;},(c:NativePlanQuestionCall)=>{c.failed=true;},(c:NativePlanQuestionCall)=>{delete c.failed;},(c:NativePlanQuestionCall)=>{delete c.answeredAt;},(c:NativePlanQuestionCall)=>{c.answeredAt='not a date';},(c:NativePlanQuestionCall)=>{c.sessionId='';},(c:NativePlanQuestionCall)=>{c.toolUseId='';},(c:NativePlanQuestionCall)=>{c.answers={};},(c:NativePlanQuestionCall)=>{c.answers={[c.questions[0]!.question]:'unoffered'};},(c:NativePlanQuestionCall)=>{c.answers!['other']='other';},(c:NativePlanQuestionCall)=>{c.unansweredQuestionIndices=[0];},(c:NativePlanQuestionCall)=>{delete c.unansweredQuestionIndices;},(c:NativePlanQuestionCall)=>{c.questions.push(structuredClone(c.questions[0]!));},(c:NativePlanQuestionCall)=>{c.questions[0]!.multiSelect=true;}])expect(classify(mutate(change))).toBe(false); + for(const f of [{...fp(),signature:'foreign:call'},{...fp(),nativeQuestionIndex:1},{...fp(),nativeCall:undefined},{...fp(),options:[...fp().options].reverse()}])expect(engFirstReviewAUQ(f)).toBe(false); + }); + test('issue, header and option identities must match without malformed explicit numbers',()=>{ + for(const c of [text(s=>s.replace('Issue 1:','Issue 01:')),text(s=>s.replace('Issue 1:','Issue 0:')),text(s=>s.replace('Issue 1:','Issue 1.2:')),text(s=>s.replace('D2 —','D02 —')),mutate(c=>{c.questions[0]!.header='Arch 2';}),mutate(c=>{c.questions[0]!.header='Scope';}),mutate(c=>{c.questions[0]!.options[0]!.label='2A: Inject AuthCache (recommended)';c.answers={[c.questions[0]!.question]:c.questions[0]!.options[0]!.label};}),mutate(c=>{c.questions[0]!.options[2]!.label=c.questions[0]!.options[1]!.label;})])expect(classify(c)).toBe(false); + }); + test('only one current metadata and assessment owner can supply the premise',()=>{ + for(const prefix of ['Source excerpt: ','Earlier review assessment: ','If approved, ','Provided this is approved, ','Historical example: '])expect(classify(text(s=>s.replace('ELI10: ','ELI10: '+prefix)))).toBe(false); + for(const prefix of ['Source: ','Earlier review assessment: ','If approved, ','Provided this is approved, '])expect(classify(text(s=>s.replace('Project/branch/task: ','Project/branch/task: '+prefix)))).toBe(false); + for(const line of ['Source excerpt:','Earlier review assessment:','Project/branch/task: a different current project','ELI10: Right now the cache is a global variable that two different services reach into and change.'])expect(classify(text(s=>s.replace('ELI10:',line+'\nELI10:')))).toBe(false); + expect(classify(text(s=>s.replace(/^Project\/branch\/task:.*\n/m,'')))).toBe(false); + expect(classify(text(s=>s.replace('a global variable that two different services reach into and change','no longer a global variable that two different services reach into and change')))).toBe(false); + }); + test('same-owner withdrawn, superseded and quoted-status claims close the question',()=>{ + for(const status of ['withdrawn','superseded','resolved','rejected','cancelled','not current','"closed"','“superseded”'])for(const subject of ['This finding','This amendment','This assessment'])expect(classify(text(s=>s+`\n${subject} is ${status}.`))).toBe(false); + expect(classify(text(s=>s+'\nThis remedy is a historical example, not the current option.'))).toBe(false); + expect(classify(text(s=>s+'\n"Earlier review assessment: This finding is withdrawn."'))).toBe(true); + }); + test('requires the named composition-root injection, removal and isolation test',()=>{ + for(const [from,to] of [['Construct one AuthCache','Construct one ForeignCache'],['AuthBroker and SessionMint constructors','AuthBroker and ForeignWriter constructors'],['AuthBroker and SessionMint constructors','AuthBroker and AuthBroker constructors'],['delete the module-level export','keep the module-level export'],['add a test that two service instances with separate caches never observe each other','tests can be added later']])expect(classify(mutate(c=>{c.questions[0]!.options[0]!.description=c.questions[0]!.options[0]!.description!.replace(from,to);}))).toBe(false); + for(const prefix of ['Source excerpt: ','If approved, ','Earlier review assessment: '])expect(classify(mutate(c=>{c.questions[0]!.options[0]!.description=prefix+c.questions[0]!.options[0]!.description;}))).toBe(false); + for(const suffix of [' This amendment is withdrawn.',' This remedy is "superseded".',' This option is not current.',' This remedy is a historical example, not the current option.'])expect(classify(mutate(c=>{c.questions[0]!.options[0]!.description+=suffix;}))).toBe(false); + }); + test('an actual opposed unchanged global and persistent risk are required',()=>{ + for(const [from,to] of [['Accept the shared global as-is.','Remove the shared global.'],['tenant leakage risk stays','tenant leakage risk is resolved']])expect(classify(mutate(c=>{c.questions[0]!.options[2]!.description=c.questions[0]!.options[2]!.description!.replace(from,to);}))).toBe(false); + for(const prefix of ['Source excerpt: ','If approved, ','Historical example: '])expect(classify(mutate(c=>{c.questions[0]!.options[2]!.description=prefix+c.questions[0]!.options[2]!.description;}))).toBe(false); + for(const suffix of [' This option is withdrawn.',' This deferral is "superseded".',' This unchanged risk is resolved.'])expect(classify(mutate(c=>{c.questions[0]!.options[2]!.description+=suffix;}))).toBe(false); + expect(classify(mutate(c=>{c.questions[0]!.options[2]!.label='1C: Start reviewing';}))).toBe(false); + }); +}); + + +test('AQ direct premise and action withdrawals supersede the earlier positive clauses',()=>{ + expect(classify(text(s=>s.replace('Project/branch/task: main','Project/branch/task: Assuming approval, main')))).toBe(false); + expect(classify(mutate(c=>{c.questions[0]!.options[0]!.description+=' Correction: do not delete the module-level export.';}))).toBe(false); + expect(classify(mutate(c=>{c.questions[0]!.options[2]!.description+=' Correction: do not accept the shared global as-is.';}))).toBe(false); + expect(classify(text(s=>s+' Correction: this cache no longer has a module-level mutable export.'))).toBe(false); +}); +}); + +describe('eng-library-hooks-aq', () => { +const fixture = fixture_eng_library_hooks_aq; +const calls=()=>structuredClone(fixture.calls) as NativePlanQuestionCall[]; +const first=()=>calls()[2]!; +const fp=(c=first())=>nativePlanCallFingerprint(c,0,true); +const classify=(c=first())=>engFirstReviewAUQ(fp(c)); +function mutate(change:(c:NativePlanQuestionCall)=>void){const c=first();change(c);return c;} +function text(change:(s:string)=>string){return mutate(c=>{const q=c.questions[0]!,answer=c.answers![q.question]!;q.question=change(q.question);c.answers={[q.question]:answer};});} +describe('AQ library-hooks choice opens batching review on its current remedy',()=>{ + test('exact twelve owned calls preserve two setup calls and ten distinct later decisions',()=>{ + let started=false;const rows=calls().map(c=>{const p=planCountQuestionPhase(fp(c),started,engStep0Boundary,engFirstReviewAUQ,engSetupAUQ);started=p.reviewStarted;return p;}); + expect(rows.map(r=>r.preReview)).toEqual([true,true,...Array(10).fill(false)]); + expect(calls().map(c=>classify(c))).toEqual([false,false,true,...Array(9).fill(false)]); + expect(classify()).toBe(true);expect(engSetupAUQ(fp())).toBe(false); + }); + test('issue numbers, option order, worker count and selected opposed choice can vary consistently',()=>{ + const c=first(),q=c.questions[0]!;q.question=q.question.replace('D3 — Architecture issue 1:','D9 — Architecture issue 4:').replaceAll('5 workers','7 workers').replace('Recommendation: 1A','Recommendation: 4A');q.header='Architecture 4'; + for(const o of q.options){o.label=o.label.replace(/^1/,'4');o.description=o.description?.replaceAll('5 copies','7 copies').replace('Five copies','Seven copies').replace('five times','seven times');}q.options.reverse(); + for(const o of q.options){c.answers={[q.question]:o.label};expect(classify(c)).toBe(true);} + }); + test('a wholly quoted archive cannot displace the current owned assessment',()=>{ + expect(classify(text(s=>s+'\n"Earlier review assessment: This finding is withdrawn."'))).toBe(true); + expect(classify(mutate(c=>{c.questions[0]!.options[0]!.description+=' "Earlier review assessment: This remedy is withdrawn."';}))).toBe(true); + }); + test('native answered-call and original menu ownership remain mandatory',()=>{ + for(const change of [(c:NativePlanQuestionCall)=>{c.answered=false;},(c:NativePlanQuestionCall)=>{c.failed=true;},(c:NativePlanQuestionCall)=>{delete c.failed;},(c:NativePlanQuestionCall)=>{delete c.answeredAt;},(c:NativePlanQuestionCall)=>{c.answeredAt='invalid';},(c:NativePlanQuestionCall)=>{c.sessionId='';},(c:NativePlanQuestionCall)=>{c.toolUseId='';},(c:NativePlanQuestionCall)=>{c.answers={};},(c:NativePlanQuestionCall)=>{c.answers={[c.questions[0]!.question]:'unoffered'};},(c:NativePlanQuestionCall)=>{c.answers!['other']='other';},(c:NativePlanQuestionCall)=>{c.unansweredQuestionIndices=[0];},(c:NativePlanQuestionCall)=>{delete c.unansweredQuestionIndices;},(c:NativePlanQuestionCall)=>{c.questions.push(structuredClone(c.questions[0]!));},(c:NativePlanQuestionCall)=>{c.questions[0]!.multiSelect=true;}])expect(classify(mutate(change))).toBe(false); + for(const f of [{...fp(),signature:'foreign:call'},{...fp(),nativeQuestionIndex:1},{...fp(),nativeCall:undefined},{...fp(),options:[...fp().options].reverse()}])expect(engFirstReviewAUQ(f)).toBe(false); + }); + test('explicit issue numbers, headers and action identities must agree',()=>{ + for(const c of [text(s=>s.replace('issue 1:','issue 01:')),text(s=>s.replace('issue 1:','issue 0:')),text(s=>s.replace('issue 1:','issue 1.2:')),text(s=>s.replace('D3 —','D03 —')),mutate(c=>{c.questions[0]!.header='Arch 2';}),mutate(c=>{c.questions[0]!.header='Scope';}),mutate(c=>{c.questions[0]!.options[0]!.label='2A: Library hooks + custom backoff fn (recommended)';c.answers={[c.questions[0]!.question]:c.questions[0]!.options[0]!.label};})])expect(classify(c)).toBe(false); + }); + test('requires unique current context and a current custom-scheduling premise',()=>{ + for(const prefix of ['Source excerpt: ','Earlier review assessment: ','If approved, ','Provided approval, ','Assuming approval, ']){ + expect(classify(text(s=>s.replace('ELI10: ','ELI10: '+prefix)))).toBe(false); + expect(classify(text(s=>s.replace('Project/branch/task: ','Project/branch/task: '+prefix)))).toBe(false); + } + for(const line of ['Source:','Earlier review assessment:','Project/branch/task: other current context','ELI10: The plan rebuilds retry scheduling by hand inside each of 5 workers.'])expect(classify(text(s=>s.replace('ELI10:',line+'\nELI10:')))).toBe(false); + expect(classify(text(s=>s.replace(/^Project\/branch\/task:.*\n/m,'')))).toBe(false); + expect(classify(text(s=>s.replace('The plan rebuilds retry scheduling','The plan no longer rebuilds retry scheduling')))).toBe(false); + }); + test('direct or quoted current withdrawal closes each owning statement',()=>{ + for(const status of ['withdrawn','superseded','resolved','"closed"','“superseded”']){ + expect(classify(text(s=>s+` This finding is ${status}.`))).toBe(false); + expect(classify(mutate(c=>{c.questions[0]!.options[0]!.description+=` This remedy is ${status}.`;}))).toBe(false); + expect(classify(mutate(c=>{c.questions[0]!.options[2]!.description+=` This option is ${status}.`;}))).toBe(false); + } + }); + test('requires a concrete library-owned retry mechanism and an isolated backoff policy',()=>{ + for(const [from,to] of [['Attempt counting, crash safety, and dashboard visibility come from the library for free.','The library could be evaluated later.'],['The backoff curve lives in one exported function','The backoff curve stays duplicated per worker']])expect(classify(mutate(c=>{c.questions[0]!.options[0]!.description=c.questions[0]!.options[0]!.description!.replace(from,to);}))).toBe(false); + for(const prefix of ['Source excerpt: ','If approved, ','Earlier review assessment: '])expect(classify(mutate(c=>{c.questions[0]!.options[0]!.description=prefix+c.questions[0]!.options[0]!.description;}))).toBe(false); + expect(classify(mutate(c=>{c.questions[0]!.options[0]!.label='1A: Start reviewing';c.answers={[c.questions[0]!.question]:c.questions[0]!.options[0]!.label};}))).toBe(false); + }); + test('unchanged scheduling must retain its current per-worker crash-safety risk',()=>{ + for(const [from,to] of [['Five copies of crash-unsafe scheduling logic','Two copies of crash-unsafe scheduling logic'],['crash-unsafe scheduling logic','crash-safe scheduling logic'],['each drifting independently','all maintained in one shared policy']])expect(classify(mutate(c=>{c.questions[0]!.options[2]!.description=c.questions[0]!.options[2]!.description!.replace(from,to);}))).toBe(false); + for(const prefix of ['Source excerpt: ','If approved, ','Historical example: '])expect(classify(mutate(c=>{c.questions[0]!.options[2]!.description=prefix+c.questions[0]!.options[2]!.description;}))).toBe(false); + expect(classify(mutate(c=>{c.questions[0]!.options[2]!.label='1C: Proceed to the next review';}))).toBe(false); + }); + test('same-owner mechanism, backoff, crash risk and premise cannot contradict their earlier claim',()=>{ + expect(classify(mutate(c=>{c.questions[0]!.options[0]!.description+='\nCorrection: the library will not own attempt counting or crash safety.';}))).toBe(false); + expect(classify(mutate(c=>{c.questions[0]!.options[0]!.description+='\nCorrection: do not preserve the exported backoff function.';}))).toBe(false); + expect(classify(mutate(c=>{c.questions[0]!.options[2]!.description+='\nCorrection: the unchanged per-worker scheduler is now crash-safe.';}))).toBe(false); + expect(classify(text(s=>s+'\nCorrection: retry scheduling no longer runs inside each worker.'))).toBe(false); + }); +}); + +describe('AY scheduler choices retain the current gap and opposed native remedies', () => { + const publicCall=(retry=false):NativePlanQuestionCall=>{ + // Minimal excerpts from the two public calls; no transcript/report corpus. + const question=retry + ? "D1 — Custom inline backoff scheduler vs the job library's built-in retry hooks\nProject/branch/task: main — background job retry framework (PLAN.md:6-8).\nELI10: The plan says (PLAN.md:6-8) to ignore it and hand-roll a scheduler inside each of the 5 workers, same shape as the library version." + : "D3 — Issue 1: custom inline scheduler per worker, or the job library's retry hook with a custom curve?\nProject/branch/task: main, PLAN.md §Architecture — background job retry framework.\nELI10: The plan writes its own \"wait, then try again\" loop inside each of the 5 workers. If that process dies mid-wait, the retry is gone and nobody knows."; + const options=retry?[ + {label:'1A) Library hooks + shared backoff fn (recommended)',description:"Register the library's retry hook in each worker, pass one shared pure backoffDelay(attempt) for the curve. Completeness 9/10."}, + {label:'1B) Custom scheduler as one shared module',description:'Roll your own, but once, with persisted retry state. You own a second job system.'}, + {label:'1C) Proceed as planned (inline in 5 workers)',description:'Keep the plan as written. Completeness 4/10. Retries die with the process; five copies drift.'}, + ]:[ + {label:'1A: Library hook + custom curve (recommended)',description:'✅ Retry state persisted by the library: survives worker crash, deploy, and restart (human: ~1 day / CC: ~20 min).'}, + {label:'1B: Custom inline scheduler as planned',description:'✅ No dependency on library hook semantics. ❌ Retry state lives in process memory: any crash mid-backoff silently drops the job; you rebuild max-attempts, dead-letter, and metrics by hand.'}, + {label:'1C: Hybrid: library hook, but custom scheduler for one worker',description:'Library persistence for 4 workers today; two retry systems remain.'}, + ]; + return {sessionId:'ay-public',toolUseId:retry?'retry':'first',questions:[{header:retry?'Architecture':'Arch 1',question,multiSelect:false,options}], + answered:true,failed:false,unansweredQuestionIndices:[],answeredAt:'2026-09-11T03:22:05.503Z',answers:{[question]:options[0]!.label}}; + }; + const edit=(retry:boolean,change:(c:NativePlanQuestionCall)=>void)=>{ + const c=publicCall(retry);change(c); + if(c.answers&&Object.keys(c.answers).length)c.answers={[c.questions[0]!.question]:c.questions[0]!.options[0]!.label}; + return c; + }; + test('both public forms start review on the same answered native choice',()=>{ + for(const retry of [false,true]){ + const c=publicCall(retry),q=c.questions[0]!; + for(const o of q.options){c.answers={[q.question]:o.label};expect(classify(c)).toBe(true);} + expect(engSetupAUQ(fp(c))).toBe(false); + } + }); + test('worker counts and native option order may vary consistently',()=>{ + for(const retry of [false,true])expect(classify(edit(retry,c=>{ + const q=c.questions[0]!;q.question=q.question.replace('5 workers','7 workers'); + q.options=q.options.map(o=>({...o,label:o.label.replace('5 workers','7 workers'),description:o.description?.replace('five copies','seven copies')})); + q.options.reverse(); + }))).toBe(true); + }); + test('same-owner native completion, metadata and menu are mandatory',()=>{ + const bad:Array<(c:NativePlanQuestionCall)=>void>=[ + c=>{c.answered=false;},c=>{c.failed=true;},c=>{delete c.answeredAt;},c=>{c.answers={};},c=>{c.unansweredQuestionIndices=[0];}, + c=>{c.questions[0]!.header='Routing';},c=>{c.questions[0]!.options[0]!.label='2A: Library hook + custom curve';}, + c=>{c.questions[0]!.question='Source excerpt:\n'+c.questions[0]!.question;}, + c=>{c.questions[0]!.question=c.questions[0]!.question.replace('ELI10: ','ELI10: If approved, ');}, + c=>{c.questions[0]!.question=c.questions[0]!.question.replace('Project/branch/task: ','Project/branch/task: If approved, ');}, + c=>{c.questions[0]!.question=c.questions[0]!.question.replace('ELI10:','> ELI10:');}, + c=>{c.questions[0]!.question+='\nELI10: The plan writes its own loop inside each of the 5 workers.';}, + c=>{c.questions[0]!.question+='\nCorrection: retry scheduling no longer runs inside each worker.';}, + ]; + for(const retry of [false,true])for(const change of bad)expect(classify(edit(retry,change))).toBe(false); + for(const retry of [false,true])expect(engFirstReviewAUQ({...fp(publicCall(retry)),signature:'foreign:call'})).toBe(false); + }); + test('the proposed library mechanism cannot borrow from another native option',()=>{ + for(const retry of [false,true]){ + const keep=retry?2:1; + for(const change of [ + (c:NativePlanQuestionCall)=>{[c.questions[0]!.options[0]!.description,c.questions[0]!.options[keep]!.description]=[c.questions[0]!.options[keep]!.description,c.questions[0]!.options[0]!.description];}, + (c:NativePlanQuestionCall)=>{c.questions[0]!.options[0]!.description='The library could be evaluated later.';}, + (c:NativePlanQuestionCall)=>{c.questions[0]!.options[0]!.description='Source excerpt: '+c.questions[0]!.options[0]!.description;}, + (c:NativePlanQuestionCall)=>{c.questions[0]!.options[0]!.description='"'+c.questions[0]!.options[0]!.description+'"';}, + (c:NativePlanQuestionCall)=>{c.questions[0]!.options[0]!.description+='\nThe library will not own persistence.';}, + (c:NativePlanQuestionCall)=>{c.questions[0]!.options[keep]!.description='Keep the current design; no retries are lost.';}, + (c:NativePlanQuestionCall)=>{c.questions[0]!.options[keep]!.description='If approved, '+c.questions[0]!.options[keep]!.description;}, + (c:NativePlanQuestionCall)=>{c.questions[0]!.options[keep]!.description+='\nThe scheduler is now crash-safe.';}, + ])expect(classify(edit(retry,change))).toBe(false); + } + expect(classify(edit(false,c=>{c.questions[0]!.options[0]!.description=c.questions[0]!.options[0]!.description!.replace('survives worker','never survives worker');}))).toBe(false); + expect(classify(edit(false,c=>{c.questions[0]!.options[1]!.description=c.questions[0]!.options[1]!.description!.replace('silently drops','never drops');}))).toBe(false); + expect(classify(edit(true,c=>{c.questions[0]!.options[2]!.description=c.questions[0]!.options[2]!.description!.replace('five copies','two copies');}))).toBe(false); + }); + test('quoted owner status and approval conditions remain current after matched outcomes',()=>{ + for(const retry of [false,true])for(const target of ['question','remedy','unchanged']){ + for(const status of ["This finding is 'withdrawn'.",'This option is conditional on approval.','If approved, proceed with this option.']){ + expect(classify(edit(retry,c=>{ + const q=c.questions[0]!; + if(target==='question')q.question+='\n'+status; + else q.options[target==='remedy'?0:retry?2:1]!.description+='\n'+status; + }))).toBe(false); + } + } + }); + test('approval clauses remain binding after option tradeoffs',()=>{ + for(const retry of [false,true])for(const option of [0,retry?2:1]){ + for(const clause of ['Assuming approval, proceed with this option.','Provided approval, keep this option.']){ + const add=(c:NativePlanQuestionCall,quoted=false)=>{c.questions[0]!.options[option]!.description+=' ❌ Additional integration effort.\n'+(quoted?'"Earlier assessment: '+clause+'"':clause);}; + expect(classify(edit(retry,c=>add(c)))).toBe(false); + expect(classify(edit(retry,c=>add(c,true)))).toBe(true); + } + } + }); + test('a crash premise cannot erase an owned approval condition',()=>{ + for(const retry of [false,true]){ + expect(classify(edit(retry,c=>{c.questions[0]!.question+='\nIf that process dies, this finding applies only if approved.';}))).toBe(false); + expect(classify(edit(retry,c=>{c.questions[0]!.question+='\n"Earlier assessment: If that process dies, this finding applies only if approved."';}))).toBe(true); + expect(classify(edit(retry,c=>{ + const q=c.questions[0]!,consequence='If the worker process crashes, the retry is gone and nobody knows.'; + q.question=retry?q.question+'\n'+consequence:q.question.replace('If that process dies mid-wait, the retry is gone and nobody knows.',consequence); + }))).toBe(true); + } + }); +}); +}); + +describe('eng-scope-y', () => { +const captured = captured_eng_scope_y; +const fresh = () => structuredClone(captured[1]!) as NativePlanQuestionCall; +const fp = (c: NativePlanQuestionCall) => nativePlanCallFingerprint(c, 0, false); +const setup = (c: NativePlanQuestionCall) => engSetupAUQ(fp(c)); +function question(c: NativePlanQuestionCall, transform: (s: string) => string) { + const q = c.questions[0]!; const answer = c.answers![q.question]!; + q.question = transform(q.question); c.answers = {[q.question]: answer}; return c; +} + +describe('Y whole-plan complexity setup decision', () => { + test('the actual accepted-complexity decision remains setup after the review boundary', () => { + expect(setup(fresh())).toBe(true); + expect(planCountQuestionPhase(fp(fresh()), true, engStep0Boundary, engFirstReviewAUQ, engSetupAUQ)) + .toEqual({preReview: true, reviewStarted: true}); + }); + + test('all seven substantive approvals and TODO obligations stay counted', () => { + let started = false; + const phases = captured.map(c => { + const call = structuredClone(c) as NativePlanQuestionCall; + const phase = planCountQuestionPhase(fp(call), started, engStep0Boundary, engFirstReviewAUQ, engSetupAUQ); + started = phase.reviewStarted; return phase.preReview; + }); + expect(phases).toEqual([true, true, true, false, false, false, false, false, false, false]); + expect(captured[8]!.questions[0]!.header).toBe('TODO: Diagrams'); + expect(captured[9]!.questions[0]!.header).toBe('TODO: Policy'); + }); + + test('either offered scope decision and reordered options remain setup', () => { + const c = fresh(); c.questions[0]!.options.reverse(); + for (const option of c.questions[0]!.options) { + c.answers = {[c.questions[0]!.question]: option.label}; expect(setup(c)).toBe(true); + } + const varied = question(fresh(), s => s.replace('4 new classes across 12 files', '6 new classes across 20 files')); + varied.questions[0]!.options[0]!.description = varied.questions[0]!.options[0]!.description.replace('4 classes across 12 files', '6 classes across 20 files'); + expect(setup(varied)).toBe(true); + }); + + test('component remedies, unfinished or conditional scope and additional work do not enter the new arm', () => { + for (const transform of [ + (s: string) => s.replace('This plan introduces', 'If this plan introduces'), + (s: string) => s.replace('This plan introduces', 'This component introduces'), + (s: string) => s.replace('This plan introduces', 'This plan does not introduce'), + (s: string) => s.replace('Recommend scope reduction before reviewing, or accept the complexity and review as-is?', 'Fix the global cache race before reviewing?'), + (s: string) => s.replace('review as-is?', 'review as-is? Also approve the cache repair.'), + (s: string) => s.replace('4 new classes', '0 new classes'), + (s: string) => s.replace('plan-eng-review-scope-challenge', 'plan-eng-review-arch-shared-cache'), + (s: string) => s.replace('plan-eng-review-scope-challenge', 'foreign-scope-challenge'), + (s: string) => s + ' ', + (s: string) => '> ' + s, + (s: string) => '```text\n' + s + '\n```', + ]) expect(setup(question(fresh(), transform))).toBe(false); + for (const index of [0, 1]) { + const c = fresh(); c.questions[0]!.options[index]!.description += ' Also implement the missing cache invalidation guard.'; + expect(setup(c)).toBe(false); + } + const mismatched = fresh(); mismatched.questions[0]!.options[0]!.description = mismatched.questions[0]!.options[0]!.description.replace('12 files', '99 files'); + expect(setup(mismatched)).toBe(false); + for (const c of captured.slice(3)) expect(setup(structuredClone(c) as NativePlanQuestionCall)).toBe(false); + }); + + test('only a matched complete native answer to the closed two-option menu qualifies', () => { + for (const mutate of [ + (c: NativePlanQuestionCall) => {c.answered = false;}, + (c: NativePlanQuestionCall) => {c.failed = true;}, + (c: NativePlanQuestionCall) => {delete c.failed;}, + (c: NativePlanQuestionCall) => {delete c.unansweredQuestionIndices;}, + (c: NativePlanQuestionCall) => {c.unansweredQuestionIndices = [0];}, + (c: NativePlanQuestionCall) => {c.questions[0]!.multiSelect = true;}, + (c: NativePlanQuestionCall) => {c.questions[0]!.header = 'Architecture';}, + (c: NativePlanQuestionCall) => {c.questions.push(structuredClone(c.questions[0]!));}, + (c: NativePlanQuestionCall) => {c.questions[0]!.options.push(structuredClone(c.questions[0]!.options[0]!));}, + (c: NativePlanQuestionCall) => {c.answers = {[c.questions[0]!.question]: 'unoffered scope decision'};}, + ]) {const c = fresh(); mutate(c); expect(setup(c)).toBe(false);} + expect(engSetupAUQ({...fp(fresh()), signature: 'foreign:call'})).toBe(false); + expect(engSetupAUQ({...fp(fresh()), nativeCall: undefined})).toBe(false); + expect(engSetupAUQ({...fp(fresh()), options: []})).toBe(false); + }); +}); +}); diff --git a/test/eng-injected-export-aq.test.ts b/test/eng-injected-export-aq.test.ts deleted file mode 100644 index a2fb19fbd..000000000 --- a/test/eng-injected-export-aq.test.ts +++ /dev/null @@ -1,66 +0,0 @@ -import {describe,expect,test} from 'bun:test'; -import {engFirstReviewAUQ,engSetupAUQ,engStep0Boundary,nativePlanCallFingerprint,planCountQuestionPhase} from './helpers/claude-pty-runner'; -import type {NativePlanQuestionCall} from './helpers/plan-count-transcript'; -import fixture from './fixtures/eng-injected-export-aq.json'; -const calls=()=>structuredClone(fixture.calls) as NativePlanQuestionCall[]; -const first=()=>calls()[1]!; -const fp=(c=first())=>nativePlanCallFingerprint(c,0,true); -const classify=(c=first())=>engFirstReviewAUQ(fp(c)); -function mutate(change:(c:NativePlanQuestionCall)=>void){const c=first();change(c);return c;} -function text(change:(s:string)=>string){return mutate(c=>{const q=c.questions[0]!,answer=c.answers![q.question]!;q.question=change(q.question);c.answers={[q.question]:answer};});} -describe('AQ current injected-export architecture decision',()=>{ - test('exact eight owned calls start review only at D2 and preserve scope first',()=>{ - let started=false;const rows=calls().map(c=>{const p=planCountQuestionPhase(fp(c),started,engStep0Boundary,engFirstReviewAUQ,engSetupAUQ);started=p.reviewStarted;return p;}); - expect(rows.map(r=>r.preReview)).toEqual([true,false,false,false,false,false,false,false]); - expect(classify()).toBe(true);expect(engSetupAUQ(fp())).toBe(false); - expect(calls().map(c=>classify(c))).toEqual([false,true,false,false,false,false,false,false]); - }); - test('consistent named actors, cache, issue and decision numbers may vary',()=>{ - const c=first(),q=c.questions[0]!; - const rename=(s:string)=>s.replaceAll('AuthCache','TokenStore').replaceAll('AuthBroker','LoginReader').replaceAll('SessionMint','SessionWriter').replace('D2 — Issue 1:','D8 — Issue 3:'); - q.question=rename(q.question);q.header='Architecture 3';for(const o of q.options){o.label=rename(o.label).replace(/^1/,'3');o.description=rename(o.description??'');} - q.options.reverse();for(const o of q.options){c.answers={[q.question]:o.label};expect(classify(c)).toBe(true);} - }); - test('the current global premise admits Today and both constructor actor orders',()=>{ - expect(classify(text(s=>s.replace('Right now the cache','Today the cache')))).toBe(true); - expect(classify(mutate(c=>{c.questions[0]!.options[0]!.description=c.questions[0]!.options[0]!.description!.replace('AuthBroker and SessionMint constructors','SessionMint and AuthBroker constructors');}))).toBe(true); - }); - test('requires complete single-question native ownership and an offered answer',()=>{ - for(const change of [(c:NativePlanQuestionCall)=>{c.answered=false;},(c:NativePlanQuestionCall)=>{c.failed=true;},(c:NativePlanQuestionCall)=>{delete c.failed;},(c:NativePlanQuestionCall)=>{delete c.answeredAt;},(c:NativePlanQuestionCall)=>{c.answeredAt='not a date';},(c:NativePlanQuestionCall)=>{c.sessionId='';},(c:NativePlanQuestionCall)=>{c.toolUseId='';},(c:NativePlanQuestionCall)=>{c.answers={};},(c:NativePlanQuestionCall)=>{c.answers={[c.questions[0]!.question]:'unoffered'};},(c:NativePlanQuestionCall)=>{c.answers!['other']='other';},(c:NativePlanQuestionCall)=>{c.unansweredQuestionIndices=[0];},(c:NativePlanQuestionCall)=>{delete c.unansweredQuestionIndices;},(c:NativePlanQuestionCall)=>{c.questions.push(structuredClone(c.questions[0]!));},(c:NativePlanQuestionCall)=>{c.questions[0]!.multiSelect=true;}])expect(classify(mutate(change))).toBe(false); - for(const f of [{...fp(),signature:'foreign:call'},{...fp(),nativeQuestionIndex:1},{...fp(),nativeCall:undefined},{...fp(),options:[...fp().options].reverse()}])expect(engFirstReviewAUQ(f)).toBe(false); - }); - test('issue, header and option identities must match without malformed explicit numbers',()=>{ - for(const c of [text(s=>s.replace('Issue 1:','Issue 01:')),text(s=>s.replace('Issue 1:','Issue 0:')),text(s=>s.replace('Issue 1:','Issue 1.2:')),text(s=>s.replace('D2 —','D02 —')),mutate(c=>{c.questions[0]!.header='Arch 2';}),mutate(c=>{c.questions[0]!.header='Scope';}),mutate(c=>{c.questions[0]!.options[0]!.label='2A: Inject AuthCache (recommended)';c.answers={[c.questions[0]!.question]:c.questions[0]!.options[0]!.label};}),mutate(c=>{c.questions[0]!.options[2]!.label=c.questions[0]!.options[1]!.label;})])expect(classify(c)).toBe(false); - }); - test('only one current metadata and assessment owner can supply the premise',()=>{ - for(const prefix of ['Source excerpt: ','Earlier review assessment: ','If approved, ','Provided this is approved, ','Historical example: '])expect(classify(text(s=>s.replace('ELI10: ','ELI10: '+prefix)))).toBe(false); - for(const prefix of ['Source: ','Earlier review assessment: ','If approved, ','Provided this is approved, '])expect(classify(text(s=>s.replace('Project/branch/task: ','Project/branch/task: '+prefix)))).toBe(false); - for(const line of ['Source excerpt:','Earlier review assessment:','Project/branch/task: a different current project','ELI10: Right now the cache is a global variable that two different services reach into and change.'])expect(classify(text(s=>s.replace('ELI10:',line+'\nELI10:')))).toBe(false); - expect(classify(text(s=>s.replace(/^Project\/branch\/task:.*\n/m,'')))).toBe(false); - expect(classify(text(s=>s.replace('a global variable that two different services reach into and change','no longer a global variable that two different services reach into and change')))).toBe(false); - }); - test('same-owner withdrawn, superseded and quoted-status claims close the question',()=>{ - for(const status of ['withdrawn','superseded','resolved','rejected','cancelled','not current','"closed"','“superseded”'])for(const subject of ['This finding','This amendment','This assessment'])expect(classify(text(s=>s+`\n${subject} is ${status}.`))).toBe(false); - expect(classify(text(s=>s+'\nThis remedy is a historical example, not the current option.'))).toBe(false); - expect(classify(text(s=>s+'\n"Earlier review assessment: This finding is withdrawn."'))).toBe(true); - }); - test('requires the named composition-root injection, removal and isolation test',()=>{ - for(const [from,to] of [['Construct one AuthCache','Construct one ForeignCache'],['AuthBroker and SessionMint constructors','AuthBroker and ForeignWriter constructors'],['AuthBroker and SessionMint constructors','AuthBroker and AuthBroker constructors'],['delete the module-level export','keep the module-level export'],['add a test that two service instances with separate caches never observe each other','tests can be added later']])expect(classify(mutate(c=>{c.questions[0]!.options[0]!.description=c.questions[0]!.options[0]!.description!.replace(from,to);}))).toBe(false); - for(const prefix of ['Source excerpt: ','If approved, ','Earlier review assessment: '])expect(classify(mutate(c=>{c.questions[0]!.options[0]!.description=prefix+c.questions[0]!.options[0]!.description;}))).toBe(false); - for(const suffix of [' This amendment is withdrawn.',' This remedy is "superseded".',' This option is not current.',' This remedy is a historical example, not the current option.'])expect(classify(mutate(c=>{c.questions[0]!.options[0]!.description+=suffix;}))).toBe(false); - }); - test('an actual opposed unchanged global and persistent risk are required',()=>{ - for(const [from,to] of [['Accept the shared global as-is.','Remove the shared global.'],['tenant leakage risk stays','tenant leakage risk is resolved']])expect(classify(mutate(c=>{c.questions[0]!.options[2]!.description=c.questions[0]!.options[2]!.description!.replace(from,to);}))).toBe(false); - for(const prefix of ['Source excerpt: ','If approved, ','Historical example: '])expect(classify(mutate(c=>{c.questions[0]!.options[2]!.description=prefix+c.questions[0]!.options[2]!.description;}))).toBe(false); - for(const suffix of [' This option is withdrawn.',' This deferral is "superseded".',' This unchanged risk is resolved.'])expect(classify(mutate(c=>{c.questions[0]!.options[2]!.description+=suffix;}))).toBe(false); - expect(classify(mutate(c=>{c.questions[0]!.options[2]!.label='1C: Start reviewing';}))).toBe(false); - }); -}); - - -test('AQ direct premise and action withdrawals supersede the earlier positive clauses',()=>{ - expect(classify(text(s=>s.replace('Project/branch/task: main','Project/branch/task: Assuming approval, main')))).toBe(false); - expect(classify(mutate(c=>{c.questions[0]!.options[0]!.description+=' Correction: do not delete the module-level export.';}))).toBe(false); - expect(classify(mutate(c=>{c.questions[0]!.options[2]!.description+=' Correction: do not accept the shared global as-is.';}))).toBe(false); - expect(classify(text(s=>s+' Correction: this cache no longer has a module-level mutable export.'))).toBe(false); -}); diff --git a/test/eng-library-hooks-aq.test.ts b/test/eng-library-hooks-aq.test.ts deleted file mode 100644 index 6b4456504..000000000 --- a/test/eng-library-hooks-aq.test.ts +++ /dev/null @@ -1,167 +0,0 @@ -import {describe,expect,test} from 'bun:test'; -import {engFirstReviewAUQ,engSetupAUQ,engStep0Boundary,nativePlanCallFingerprint,planCountQuestionPhase} from './helpers/claude-pty-runner'; -import type {NativePlanQuestionCall} from './helpers/plan-count-transcript'; -import fixture from './fixtures/eng-library-hooks-aq.json'; -const calls=()=>structuredClone(fixture.calls) as NativePlanQuestionCall[]; -const first=()=>calls()[2]!; -const fp=(c=first())=>nativePlanCallFingerprint(c,0,true); -const classify=(c=first())=>engFirstReviewAUQ(fp(c)); -function mutate(change:(c:NativePlanQuestionCall)=>void){const c=first();change(c);return c;} -function text(change:(s:string)=>string){return mutate(c=>{const q=c.questions[0]!,answer=c.answers![q.question]!;q.question=change(q.question);c.answers={[q.question]:answer};});} -describe('AQ library-hooks choice opens batching review on its current remedy',()=>{ - test('exact twelve owned calls preserve two setup calls and ten distinct later decisions',()=>{ - let started=false;const rows=calls().map(c=>{const p=planCountQuestionPhase(fp(c),started,engStep0Boundary,engFirstReviewAUQ,engSetupAUQ);started=p.reviewStarted;return p;}); - expect(rows.map(r=>r.preReview)).toEqual([true,true,...Array(10).fill(false)]); - expect(calls().map(c=>classify(c))).toEqual([false,false,true,...Array(9).fill(false)]); - expect(classify()).toBe(true);expect(engSetupAUQ(fp())).toBe(false); - }); - test('issue numbers, option order, worker count and selected opposed choice can vary consistently',()=>{ - const c=first(),q=c.questions[0]!;q.question=q.question.replace('D3 — Architecture issue 1:','D9 — Architecture issue 4:').replaceAll('5 workers','7 workers').replace('Recommendation: 1A','Recommendation: 4A');q.header='Architecture 4'; - for(const o of q.options){o.label=o.label.replace(/^1/,'4');o.description=o.description?.replaceAll('5 copies','7 copies').replace('Five copies','Seven copies').replace('five times','seven times');}q.options.reverse(); - for(const o of q.options){c.answers={[q.question]:o.label};expect(classify(c)).toBe(true);} - }); - test('a wholly quoted archive cannot displace the current owned assessment',()=>{ - expect(classify(text(s=>s+'\n"Earlier review assessment: This finding is withdrawn."'))).toBe(true); - expect(classify(mutate(c=>{c.questions[0]!.options[0]!.description+=' "Earlier review assessment: This remedy is withdrawn."';}))).toBe(true); - }); - test('native answered-call and original menu ownership remain mandatory',()=>{ - for(const change of [(c:NativePlanQuestionCall)=>{c.answered=false;},(c:NativePlanQuestionCall)=>{c.failed=true;},(c:NativePlanQuestionCall)=>{delete c.failed;},(c:NativePlanQuestionCall)=>{delete c.answeredAt;},(c:NativePlanQuestionCall)=>{c.answeredAt='invalid';},(c:NativePlanQuestionCall)=>{c.sessionId='';},(c:NativePlanQuestionCall)=>{c.toolUseId='';},(c:NativePlanQuestionCall)=>{c.answers={};},(c:NativePlanQuestionCall)=>{c.answers={[c.questions[0]!.question]:'unoffered'};},(c:NativePlanQuestionCall)=>{c.answers!['other']='other';},(c:NativePlanQuestionCall)=>{c.unansweredQuestionIndices=[0];},(c:NativePlanQuestionCall)=>{delete c.unansweredQuestionIndices;},(c:NativePlanQuestionCall)=>{c.questions.push(structuredClone(c.questions[0]!));},(c:NativePlanQuestionCall)=>{c.questions[0]!.multiSelect=true;}])expect(classify(mutate(change))).toBe(false); - for(const f of [{...fp(),signature:'foreign:call'},{...fp(),nativeQuestionIndex:1},{...fp(),nativeCall:undefined},{...fp(),options:[...fp().options].reverse()}])expect(engFirstReviewAUQ(f)).toBe(false); - }); - test('explicit issue numbers, headers and action identities must agree',()=>{ - for(const c of [text(s=>s.replace('issue 1:','issue 01:')),text(s=>s.replace('issue 1:','issue 0:')),text(s=>s.replace('issue 1:','issue 1.2:')),text(s=>s.replace('D3 —','D03 —')),mutate(c=>{c.questions[0]!.header='Arch 2';}),mutate(c=>{c.questions[0]!.header='Scope';}),mutate(c=>{c.questions[0]!.options[0]!.label='2A: Library hooks + custom backoff fn (recommended)';c.answers={[c.questions[0]!.question]:c.questions[0]!.options[0]!.label};})])expect(classify(c)).toBe(false); - }); - test('requires unique current context and a current custom-scheduling premise',()=>{ - for(const prefix of ['Source excerpt: ','Earlier review assessment: ','If approved, ','Provided approval, ','Assuming approval, ']){ - expect(classify(text(s=>s.replace('ELI10: ','ELI10: '+prefix)))).toBe(false); - expect(classify(text(s=>s.replace('Project/branch/task: ','Project/branch/task: '+prefix)))).toBe(false); - } - for(const line of ['Source:','Earlier review assessment:','Project/branch/task: other current context','ELI10: The plan rebuilds retry scheduling by hand inside each of 5 workers.'])expect(classify(text(s=>s.replace('ELI10:',line+'\nELI10:')))).toBe(false); - expect(classify(text(s=>s.replace(/^Project\/branch\/task:.*\n/m,'')))).toBe(false); - expect(classify(text(s=>s.replace('The plan rebuilds retry scheduling','The plan no longer rebuilds retry scheduling')))).toBe(false); - }); - test('direct or quoted current withdrawal closes each owning statement',()=>{ - for(const status of ['withdrawn','superseded','resolved','"closed"','“superseded”']){ - expect(classify(text(s=>s+` This finding is ${status}.`))).toBe(false); - expect(classify(mutate(c=>{c.questions[0]!.options[0]!.description+=` This remedy is ${status}.`;}))).toBe(false); - expect(classify(mutate(c=>{c.questions[0]!.options[2]!.description+=` This option is ${status}.`;}))).toBe(false); - } - }); - test('requires a concrete library-owned retry mechanism and an isolated backoff policy',()=>{ - for(const [from,to] of [['Attempt counting, crash safety, and dashboard visibility come from the library for free.','The library could be evaluated later.'],['The backoff curve lives in one exported function','The backoff curve stays duplicated per worker']])expect(classify(mutate(c=>{c.questions[0]!.options[0]!.description=c.questions[0]!.options[0]!.description!.replace(from,to);}))).toBe(false); - for(const prefix of ['Source excerpt: ','If approved, ','Earlier review assessment: '])expect(classify(mutate(c=>{c.questions[0]!.options[0]!.description=prefix+c.questions[0]!.options[0]!.description;}))).toBe(false); - expect(classify(mutate(c=>{c.questions[0]!.options[0]!.label='1A: Start reviewing';c.answers={[c.questions[0]!.question]:c.questions[0]!.options[0]!.label};}))).toBe(false); - }); - test('unchanged scheduling must retain its current per-worker crash-safety risk',()=>{ - for(const [from,to] of [['Five copies of crash-unsafe scheduling logic','Two copies of crash-unsafe scheduling logic'],['crash-unsafe scheduling logic','crash-safe scheduling logic'],['each drifting independently','all maintained in one shared policy']])expect(classify(mutate(c=>{c.questions[0]!.options[2]!.description=c.questions[0]!.options[2]!.description!.replace(from,to);}))).toBe(false); - for(const prefix of ['Source excerpt: ','If approved, ','Historical example: '])expect(classify(mutate(c=>{c.questions[0]!.options[2]!.description=prefix+c.questions[0]!.options[2]!.description;}))).toBe(false); - expect(classify(mutate(c=>{c.questions[0]!.options[2]!.label='1C: Proceed to the next review';}))).toBe(false); - }); - test('same-owner mechanism, backoff, crash risk and premise cannot contradict their earlier claim',()=>{ - expect(classify(mutate(c=>{c.questions[0]!.options[0]!.description+='\nCorrection: the library will not own attempt counting or crash safety.';}))).toBe(false); - expect(classify(mutate(c=>{c.questions[0]!.options[0]!.description+='\nCorrection: do not preserve the exported backoff function.';}))).toBe(false); - expect(classify(mutate(c=>{c.questions[0]!.options[2]!.description+='\nCorrection: the unchanged per-worker scheduler is now crash-safe.';}))).toBe(false); - expect(classify(text(s=>s+'\nCorrection: retry scheduling no longer runs inside each worker.'))).toBe(false); - }); -}); - -describe('AY scheduler choices retain the current gap and opposed native remedies', () => { - const publicCall=(retry=false):NativePlanQuestionCall=>{ - // Minimal excerpts from the two public calls; no transcript/report corpus. - const question=retry - ? "D1 — Custom inline backoff scheduler vs the job library's built-in retry hooks\nProject/branch/task: main — background job retry framework (PLAN.md:6-8).\nELI10: The plan says (PLAN.md:6-8) to ignore it and hand-roll a scheduler inside each of the 5 workers, same shape as the library version." - : "D3 — Issue 1: custom inline scheduler per worker, or the job library's retry hook with a custom curve?\nProject/branch/task: main, PLAN.md §Architecture — background job retry framework.\nELI10: The plan writes its own \"wait, then try again\" loop inside each of the 5 workers. If that process dies mid-wait, the retry is gone and nobody knows."; - const options=retry?[ - {label:'1A) Library hooks + shared backoff fn (recommended)',description:"Register the library's retry hook in each worker, pass one shared pure backoffDelay(attempt) for the curve. Completeness 9/10."}, - {label:'1B) Custom scheduler as one shared module',description:'Roll your own, but once, with persisted retry state. You own a second job system.'}, - {label:'1C) Proceed as planned (inline in 5 workers)',description:'Keep the plan as written. Completeness 4/10. Retries die with the process; five copies drift.'}, - ]:[ - {label:'1A: Library hook + custom curve (recommended)',description:'✅ Retry state persisted by the library: survives worker crash, deploy, and restart (human: ~1 day / CC: ~20 min).'}, - {label:'1B: Custom inline scheduler as planned',description:'✅ No dependency on library hook semantics. ❌ Retry state lives in process memory: any crash mid-backoff silently drops the job; you rebuild max-attempts, dead-letter, and metrics by hand.'}, - {label:'1C: Hybrid: library hook, but custom scheduler for one worker',description:'Library persistence for 4 workers today; two retry systems remain.'}, - ]; - return {sessionId:'ay-public',toolUseId:retry?'retry':'first',questions:[{header:retry?'Architecture':'Arch 1',question,multiSelect:false,options}], - answered:true,failed:false,unansweredQuestionIndices:[],answeredAt:'2026-09-11T03:22:05.503Z',answers:{[question]:options[0]!.label}}; - }; - const edit=(retry:boolean,change:(c:NativePlanQuestionCall)=>void)=>{ - const c=publicCall(retry);change(c); - if(c.answers&&Object.keys(c.answers).length)c.answers={[c.questions[0]!.question]:c.questions[0]!.options[0]!.label}; - return c; - }; - test('both public forms start review on the same answered native choice',()=>{ - for(const retry of [false,true]){ - const c=publicCall(retry),q=c.questions[0]!; - for(const o of q.options){c.answers={[q.question]:o.label};expect(classify(c)).toBe(true);} - expect(engSetupAUQ(fp(c))).toBe(false); - } - }); - test('worker counts and native option order may vary consistently',()=>{ - for(const retry of [false,true])expect(classify(edit(retry,c=>{ - const q=c.questions[0]!;q.question=q.question.replace('5 workers','7 workers'); - q.options=q.options.map(o=>({...o,label:o.label.replace('5 workers','7 workers'),description:o.description?.replace('five copies','seven copies')})); - q.options.reverse(); - }))).toBe(true); - }); - test('same-owner native completion, metadata and menu are mandatory',()=>{ - const bad:Array<(c:NativePlanQuestionCall)=>void>=[ - c=>{c.answered=false;},c=>{c.failed=true;},c=>{delete c.answeredAt;},c=>{c.answers={};},c=>{c.unansweredQuestionIndices=[0];}, - c=>{c.questions[0]!.header='Routing';},c=>{c.questions[0]!.options[0]!.label='2A: Library hook + custom curve';}, - c=>{c.questions[0]!.question='Source excerpt:\n'+c.questions[0]!.question;}, - c=>{c.questions[0]!.question=c.questions[0]!.question.replace('ELI10: ','ELI10: If approved, ');}, - c=>{c.questions[0]!.question=c.questions[0]!.question.replace('Project/branch/task: ','Project/branch/task: If approved, ');}, - c=>{c.questions[0]!.question=c.questions[0]!.question.replace('ELI10:','> ELI10:');}, - c=>{c.questions[0]!.question+='\nELI10: The plan writes its own loop inside each of the 5 workers.';}, - c=>{c.questions[0]!.question+='\nCorrection: retry scheduling no longer runs inside each worker.';}, - ]; - for(const retry of [false,true])for(const change of bad)expect(classify(edit(retry,change))).toBe(false); - for(const retry of [false,true])expect(engFirstReviewAUQ({...fp(publicCall(retry)),signature:'foreign:call'})).toBe(false); - }); - test('the proposed library mechanism cannot borrow from another native option',()=>{ - for(const retry of [false,true]){ - const keep=retry?2:1; - for(const change of [ - (c:NativePlanQuestionCall)=>{[c.questions[0]!.options[0]!.description,c.questions[0]!.options[keep]!.description]=[c.questions[0]!.options[keep]!.description,c.questions[0]!.options[0]!.description];}, - (c:NativePlanQuestionCall)=>{c.questions[0]!.options[0]!.description='The library could be evaluated later.';}, - (c:NativePlanQuestionCall)=>{c.questions[0]!.options[0]!.description='Source excerpt: '+c.questions[0]!.options[0]!.description;}, - (c:NativePlanQuestionCall)=>{c.questions[0]!.options[0]!.description='"'+c.questions[0]!.options[0]!.description+'"';}, - (c:NativePlanQuestionCall)=>{c.questions[0]!.options[0]!.description+='\nThe library will not own persistence.';}, - (c:NativePlanQuestionCall)=>{c.questions[0]!.options[keep]!.description='Keep the current design; no retries are lost.';}, - (c:NativePlanQuestionCall)=>{c.questions[0]!.options[keep]!.description='If approved, '+c.questions[0]!.options[keep]!.description;}, - (c:NativePlanQuestionCall)=>{c.questions[0]!.options[keep]!.description+='\nThe scheduler is now crash-safe.';}, - ])expect(classify(edit(retry,change))).toBe(false); - } - expect(classify(edit(false,c=>{c.questions[0]!.options[0]!.description=c.questions[0]!.options[0]!.description!.replace('survives worker','never survives worker');}))).toBe(false); - expect(classify(edit(false,c=>{c.questions[0]!.options[1]!.description=c.questions[0]!.options[1]!.description!.replace('silently drops','never drops');}))).toBe(false); - expect(classify(edit(true,c=>{c.questions[0]!.options[2]!.description=c.questions[0]!.options[2]!.description!.replace('five copies','two copies');}))).toBe(false); - }); - test('quoted owner status and approval conditions remain current after matched outcomes',()=>{ - for(const retry of [false,true])for(const target of ['question','remedy','unchanged']){ - for(const status of ["This finding is 'withdrawn'.",'This option is conditional on approval.','If approved, proceed with this option.']){ - expect(classify(edit(retry,c=>{ - const q=c.questions[0]!; - if(target==='question')q.question+='\n'+status; - else q.options[target==='remedy'?0:retry?2:1]!.description+='\n'+status; - }))).toBe(false); - } - } - }); - test('approval clauses remain binding after option tradeoffs',()=>{ - for(const retry of [false,true])for(const option of [0,retry?2:1]){ - for(const clause of ['Assuming approval, proceed with this option.','Provided approval, keep this option.']){ - const add=(c:NativePlanQuestionCall,quoted=false)=>{c.questions[0]!.options[option]!.description+=' ❌ Additional integration effort.\n'+(quoted?'"Earlier assessment: '+clause+'"':clause);}; - expect(classify(edit(retry,c=>add(c)))).toBe(false); - expect(classify(edit(retry,c=>add(c,true)))).toBe(true); - } - } - }); - test('a crash premise cannot erase an owned approval condition',()=>{ - for(const retry of [false,true]){ - expect(classify(edit(retry,c=>{c.questions[0]!.question+='\nIf that process dies, this finding applies only if approved.';}))).toBe(false); - expect(classify(edit(retry,c=>{c.questions[0]!.question+='\n"Earlier assessment: If that process dies, this finding applies only if approved."';}))).toBe(true); - expect(classify(edit(retry,c=>{ - const q=c.questions[0]!,consequence='If the worker process crashes, the retry is gone and nobody knows.'; - q.question=retry?q.question+'\n'+consequence:q.question.replace('If that process dies mid-wait, the retry is gone and nobody knows.',consequence); - }))).toBe(true); - } - }); -}); diff --git a/test/eng-next-handoff-ah.test.ts b/test/eng-next-handoff-ah.test.ts deleted file mode 100644 index 0401c9e50..000000000 --- a/test/eng-next-handoff-ah.test.ts +++ /dev/null @@ -1,64 +0,0 @@ -import { expect, test } from 'bun:test'; -import fs from 'node:fs'; -import os from 'node:os'; -import path from 'node:path'; -import actual from './fixtures/eng-next-handoff-ah.json'; -import { hasNativePlanTerminal } from './helpers/claude-pty-runner'; -import type { PlanCountTranscript } from './helpers/plan-count-transcript'; -import { isCurrentPlanApprovalScreen } from './helpers/plan-count-pending-exit'; - -test('exact final exit/report replay retains all freshness, identity and answer gates', () => { - const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'gstack-eng-next-ah-')); - const file = path.join(dir, 'reviewed.md'); - const now = Date.now; - try { - fs.writeFileSync(file, actual.plan); - fs.utimesSync(file, actual.source.stat.mtimeMs / 1000, actual.source.stat.mtimeMs / 1000); - Date.now = () => Date.parse(actual.captureAt); - const t = structuredClone(actual.transcript) as PlanCountTranscript; - const id = actual.fingerprint.signature; - const admin = new Set([id]); - const check = (v = t, a = admin) => hasNativePlanTerminal(v, file, actual.startedAt, 'plan_ready', a); - expect(isCurrentPlanApprovalScreen(actual.screen)).toBe(true); - expect(check()).toBe(true); - expect(check(t, new Set())).toBe(false); - expect(check(t, new Set(['foreign:call']))).toBe(false); - for (const mutate of [ - (v: PlanCountTranscript) => { v.planReadyRequests = []; }, - (v: PlanCountTranscript) => { v.planReadyRequests!.at(-1)!.failed = true; }, - (v: PlanCountTranscript) => { v.planReadyRequests!.at(-1)!.sessionId = 'foreign'; }, - (v: PlanCountTranscript) => { v.planReadyRequests!.at(-1)!.timestamp = '2026-09-10T03:29:40.000Z'; }, - (v: PlanCountTranscript) => { v.planReadyRequests!.at(-1)!.timestamp = new Date(Date.now() + 1).toISOString(); }, - (v: PlanCountTranscript) => { v.calls.at(-1)!.answered = false; }, - (v: PlanCountTranscript) => { v.calls.at(-2)!.answeredAt = '2026-09-10T03:29:00.000Z'; }, - ]) { const v = structuredClone(t); mutate(v); expect(check(v)).toBe(false); } - fs.writeFileSync(file, actual.plan.replace('NO UNRESOLVED DECISIONS', 'Report still pending')); - fs.utimesSync(file, actual.source.stat.mtimeMs / 1000, actual.source.stat.mtimeMs / 1000); - expect(check()).toBe(false); - } finally { Date.now = now; fs.rmSync(dir, { recursive: true, force: true }); } -}); - -const b176 = actual.sourceBoundB176; -const recorded = () => structuredClone(b176.transcript) as PlanCountTranscript; -test('actual pending ExitPlanMode needs the classified recap plus the unchanged fresh report and native gates',()=>{ - const dir=fs.mkdtempSync(path.join(os.tmpdir(),'eng-b176-terminal-')),file=path.join(dir,'report.md'),now=Date.now; - try{ - Date.now=()=>Date.parse(b176.capturedAt);fs.writeFileSync(file,b176.plan);fs.utimesSync(file,b176.sourceReport.mtimeMs/1000,b176.sourceReport.mtimeMs/1000); - const t=recorded(),signature=b176.fingerprint.signature; - const admin=new Set([signature]); - const check=(transcript=t,administrative=admin)=>hasNativePlanTerminal(transcript,file,b176.startedAt,'plan_ready',administrative); - expect(isCurrentPlanApprovalScreen(b176.screen)).toBe(true); - expect(check()).toBe(true);expect(check(t,new Set())).toBe(false);expect(check(t,new Set(['foreign:call']))).toBe(false); - for(const mutate of [ - (v:PlanCountTranscript)=>{v.planReadyRequests=[];}, - (v:PlanCountTranscript)=>{v.planReadyRequests![0]!.failed=true;}, - (v:PlanCountTranscript)=>{v.planReadyRequests![0]!.sessionId='foreign';}, - (v:PlanCountTranscript)=>{v.planReadyRequests![0]!.timestamp=new Date(Date.now()+1).toISOString();}, - (v:PlanCountTranscript)=>{v.planReadyRequests![0]!.timestamp=v.calls.at(-1)!.answeredAt!;}, - (v:PlanCountTranscript)=>{v.calls.at(-1)!.answered=false;}, - (v:PlanCountTranscript)=>{v.calls.at(-2)!.answeredAt=new Date(b176.sourceReport.mtimeMs+1).toISOString();}, - ]){const v=recorded();mutate(v);expect(check(v)).toBe(false);} - fs.utimesSync(file,(b176.startedAt-1)/1000,(b176.startedAt-1)/1000);expect(check()).toBe(false); - fs.writeFileSync(file,b176.plan.replace('NO UNRESOLVED DECISIONS','PENDING'));fs.utimesSync(file,b176.sourceReport.mtimeMs/1000,b176.sourceReport.mtimeMs/1000);expect(check()).toBe(false); - }finally{Date.now=now;fs.rmSync(dir,{recursive:true,force:true});} -}); diff --git a/test/eng-option-b-scope-al.test.ts b/test/eng-option-b-scope-al.test.ts deleted file mode 100644 index d22de25f2..000000000 --- a/test/eng-option-b-scope-al.test.ts +++ /dev/null @@ -1,118 +0,0 @@ -import { expect, test } from 'bun:test'; -import { nativeSeededPlanSelection } from './helpers/plan-scope-selection'; -import type { NativePublicToolEvent, PlanCountTranscript } from './helpers/plan-count-transcript'; -import { E2E_TOUCHFILES, selectTests } from './helpers/touchfiles'; -import fixture from './fixtures/eng-option-b-scope-al.json'; - -const actualInput = (attempt = 1) => structuredClone(fixture.attempts[attempt]!.projection); -const input = () => { - const p = actualInput(); - // Mutate the named declaration alone; the earlier spoken introduction is - // independently valid and remains present in the exact replays below. - p.transcript.assistantMessages = p.transcript.assistantMessages.filter(m => m.text !== "I'll run the eng review skill on this draft plan."); - return p; -}; -type Input = ReturnType; -const verdict = (p = input()) => nativeSeededPlanSelection(p.transcript as PlanCountTranscript, p.tools as NativePublicToolEvent[], p.opts); -const declaration = (p: Input) => p.transcript.assistantMessages.find(m => m.text.startsWith("I've selected option B,"))!; - -test('both named retry and fresh unique-draft first introduction bind; original outcomes stay intact', () => { - expect(fixture.attempts.map(a => a.rawScopeGateAutoSelectObserved)).toEqual([false, false]); - expect(verdict(actualInput(0))).toBe(true); - expect(verdict(actualInput(1))).toBe(true); -}); - -test('an unnamed option-B notice supplies no selection without a draft introduction', () => { - const p = input(); p.transcript.assistantMessages = p.transcript.assistantMessages.filter(m => m !== declaration(p)); - expect(verdict(p)).toBe(false); - for (const replacement of ['option A,', 'option C,', 'option B if approved,', 'option B, possibly']) { - const p = input(); declaration(p).text = declaration(p).text.replace('option B,', replacement); expect(verdict(p)).toBe(false); - } - for (const target of ['Unrelated draft', 'branch diff']) { - const p = input(); declaration(p).text = declaration(p).text.replace('Parallelize unit tests', target); expect(verdict(p)).toBe(false); - } -}); - -test('equivalent current wording and consistently renamed title preserve selection', () => { - for (const change of [ - (s: string) => s.replace("I've", 'I have'), - (s: string) => s.replace('Next I', 'Next, I'), - (s: string) => s.replace('reviewing the pasted', 'to review the pasted'), - (s: string) => s.replace(/\. Next.*$/, '.'), - (s: string) => s.replace('Design Doc Check, brain context, and context recovery, along with the Aside probe', 'audit for DESIGN.md'), - ]) { const p = input(); declaration(p).text = change(declaration(p).text); expect(verdict(p)).toBe(true); } - const p = input(); p.opts.seed = p.opts.seed.replace('Parallelize unit tests', 'Build cache invalidation'); - declaration(p).text = declaration(p).text.replace('Parallelize unit tests', 'Build cache invalidation'); expect(verdict(p)).toBe(true); -}); - -test('quoted, source, hypothetical, historical and conditional first lines cannot select', () => { - for (const prefix of ['> ', ' ', '\t', '```\n', 'Source excerpt:\n', 'Historical example only.\n', 'The following is a hypothetical example. ', 'If approved, ', '"']) { - const p = input(); declaration(p).text = prefix + declaration(p).text; expect(verdict(p)).toBe(false); - } -}); - -test('the same successful post-command Skill completion and current native session are required', () => { - for (const change of [ - (p: Input) => { p.opts.sessionId = 'foreign'; }, - (p: Input) => { p.transcript.status = 'unavailable'; }, - (p: Input) => { p.tools = []; }, - (p: Input) => { p.tools[0]!.input!.skill = 'plan-design-review'; }, - (p: Input) => { p.tools[1]!.isError = true; }, - (p: Input) => { p.tools[1]!.sessionId = 'foreign'; }, - (p: Input) => { p.tools[1]!.toolUseId = 'unrelated'; }, - (p: Input) => { p.opts.commandStartedAt = Date.parse(p.tools[0]!.timestamp) + 1; }, - (p: Input) => { declaration(p).timestamp = new Date(p.opts.commandStartedAt - 1).toISOString(); }, - (p: Input) => { declaration(p).sessionId = 'foreign'; }, - (p: Input) => { p.tools.push(structuredClone(p.tools[1]!)); }, - ]) { const p = input(); change(p); expect(verdict(p)).toBe(false); } -}); - -test('conditional, questioning and replacement continuations cannot borrow a completed selection', () => { - for (const change of [ - (s: string) => s.replace('draft plan.', 'draft plan if approved.'), - (s: string) => s.replace('Next I', 'If approved, I'), - (s: string) => s.replace('Aside probe.', 'Aside probe?'), - (s: string) => s.replace('Design Doc Check', 'branch diff review instead'), - (s: string) => s.replace('Design Doc Check', 'unrelated work'), - ]) { const p = input(); declaration(p).text = change(declaration(p).text); expect(verdict(p)).toBe(false); } -}); - -test('same-message or later owned withdrawals and target changes defeat the declaration', () => { - for (const correction of ['Correction: this selection is withdrawn.', 'This declaration has been retracted.', 'The selected target is now the branch diff.']) { - for (const placement of ['same-line', 'same-message', 'later']) { - const p = input(), m = declaration(p); - if (placement === 'later') p.transcript.assistantMessages.push({ ...m, timestamp: new Date(Date.parse(m.timestamp) + 1000).toISOString(), text: correction }); - else m.text += (placement === 'same-line' ? ' ' : '\n') + correction; - expect(verdict(p)).toBe(false); - } - } -}); - -test('foreign, historical and literal corrections do not retract a current named selection', () => { - for (const text of ['> This selection is withdrawn.', 'Source excerpt:\nThis selection is withdrawn.', 'A prior assistant said "This selection is withdrawn."', 'The verification suite is withdrawn.', 'Is this selection withdrawn?']) { - const p = input(), m = declaration(p); p.transcript.assistantMessages.push({ ...m, timestamp: new Date(Date.parse(m.timestamp) + 1000).toISOString(), text }); expect(verdict(p)).toBe(true); - } - const p = input(), m = declaration(p); p.transcript.assistantMessages.push({ ...m, sessionId: 'foreign', text: 'This selection is withdrawn.' }); expect(verdict(p)).toBe(true); -}); - -test('a later explicit reselection follows the existing currentness rule', () => { - const p = input(), m = declaration(p); p.transcript.assistantMessages.push({ ...m, timestamp: new Date(Date.parse(m.timestamp) + 1000).toISOString(), text: 'This selection is withdrawn.' }); expect(verdict(p)).toBe(false); - p.transcript.assistantMessages.push({ ...m, timestamp: new Date(Date.parse(m.timestamp) + 2000).toISOString() }); expect(verdict(p)).toBe(true); -}); - -test('both new dependencies select exactly the existing five scope observers', () => { - const expected = selectTests(['test/helpers/plan-scope-selection.ts'], E2E_TOUCHFILES, []).selected; - expect(expected).toHaveLength(5); - for (const path of ['test/eng-option-b-scope-al.test.ts', 'test/fixtures/eng-option-b-scope-al.json']) expect(selectTests([path], E2E_TOUCHFILES, []).selected).toEqual(expected); -}); - -for (const owner of [ - 'plan-ceo-review-plan-mode', 'plan-eng-review-plan-mode', - 'plan-design-review-plan-mode', 'plan-devex-review-plan-mode', 'plan-mode-no-op', -]) test(`scope dependency registration is dense for ${owner}`, () => { - const paths = E2E_TOUCHFILES[owner]!; - for (let index = 0; index < paths.length; index++) { - expect(Object.hasOwn(paths, index)).toBe(true); - expect(typeof paths[index]).toBe('string'); - } -}); diff --git a/test/eng-scope-y.test.ts b/test/eng-scope-y.test.ts deleted file mode 100644 index 9b88e344a..000000000 --- a/test/eng-scope-y.test.ts +++ /dev/null @@ -1,83 +0,0 @@ -import { describe, expect, test } from 'bun:test'; -import captured from './fixtures/eng-scope-y-calls.json'; -import { engFirstReviewAUQ, engSetupAUQ, engStep0Boundary, nativePlanCallFingerprint, planCountQuestionPhase } from './helpers/claude-pty-runner'; -import type { NativePlanQuestionCall } from './helpers/plan-count-transcript'; - -const fresh = () => structuredClone(captured[1]!) as NativePlanQuestionCall; -const fp = (c: NativePlanQuestionCall) => nativePlanCallFingerprint(c, 0, false); -const setup = (c: NativePlanQuestionCall) => engSetupAUQ(fp(c)); -function question(c: NativePlanQuestionCall, transform: (s: string) => string) { - const q = c.questions[0]!; const answer = c.answers![q.question]!; - q.question = transform(q.question); c.answers = {[q.question]: answer}; return c; -} - -describe('Y whole-plan complexity setup decision', () => { - test('the actual accepted-complexity decision remains setup after the review boundary', () => { - expect(setup(fresh())).toBe(true); - expect(planCountQuestionPhase(fp(fresh()), true, engStep0Boundary, engFirstReviewAUQ, engSetupAUQ)) - .toEqual({preReview: true, reviewStarted: true}); - }); - - test('all seven substantive approvals and TODO obligations stay counted', () => { - let started = false; - const phases = captured.map(c => { - const call = structuredClone(c) as NativePlanQuestionCall; - const phase = planCountQuestionPhase(fp(call), started, engStep0Boundary, engFirstReviewAUQ, engSetupAUQ); - started = phase.reviewStarted; return phase.preReview; - }); - expect(phases).toEqual([true, true, true, false, false, false, false, false, false, false]); - expect(captured[8]!.questions[0]!.header).toBe('TODO: Diagrams'); - expect(captured[9]!.questions[0]!.header).toBe('TODO: Policy'); - }); - - test('either offered scope decision and reordered options remain setup', () => { - const c = fresh(); c.questions[0]!.options.reverse(); - for (const option of c.questions[0]!.options) { - c.answers = {[c.questions[0]!.question]: option.label}; expect(setup(c)).toBe(true); - } - const varied = question(fresh(), s => s.replace('4 new classes across 12 files', '6 new classes across 20 files')); - varied.questions[0]!.options[0]!.description = varied.questions[0]!.options[0]!.description.replace('4 classes across 12 files', '6 classes across 20 files'); - expect(setup(varied)).toBe(true); - }); - - test('component remedies, unfinished or conditional scope and additional work do not enter the new arm', () => { - for (const transform of [ - (s: string) => s.replace('This plan introduces', 'If this plan introduces'), - (s: string) => s.replace('This plan introduces', 'This component introduces'), - (s: string) => s.replace('This plan introduces', 'This plan does not introduce'), - (s: string) => s.replace('Recommend scope reduction before reviewing, or accept the complexity and review as-is?', 'Fix the global cache race before reviewing?'), - (s: string) => s.replace('review as-is?', 'review as-is? Also approve the cache repair.'), - (s: string) => s.replace('4 new classes', '0 new classes'), - (s: string) => s.replace('plan-eng-review-scope-challenge', 'plan-eng-review-arch-shared-cache'), - (s: string) => s.replace('plan-eng-review-scope-challenge', 'foreign-scope-challenge'), - (s: string) => s + ' ', - (s: string) => '> ' + s, - (s: string) => '```text\n' + s + '\n```', - ]) expect(setup(question(fresh(), transform))).toBe(false); - for (const index of [0, 1]) { - const c = fresh(); c.questions[0]!.options[index]!.description += ' Also implement the missing cache invalidation guard.'; - expect(setup(c)).toBe(false); - } - const mismatched = fresh(); mismatched.questions[0]!.options[0]!.description = mismatched.questions[0]!.options[0]!.description.replace('12 files', '99 files'); - expect(setup(mismatched)).toBe(false); - for (const c of captured.slice(3)) expect(setup(structuredClone(c) as NativePlanQuestionCall)).toBe(false); - }); - - test('only a matched complete native answer to the closed two-option menu qualifies', () => { - for (const mutate of [ - (c: NativePlanQuestionCall) => {c.answered = false;}, - (c: NativePlanQuestionCall) => {c.failed = true;}, - (c: NativePlanQuestionCall) => {delete c.failed;}, - (c: NativePlanQuestionCall) => {delete c.unansweredQuestionIndices;}, - (c: NativePlanQuestionCall) => {c.unansweredQuestionIndices = [0];}, - (c: NativePlanQuestionCall) => {c.questions[0]!.multiSelect = true;}, - (c: NativePlanQuestionCall) => {c.questions[0]!.header = 'Architecture';}, - (c: NativePlanQuestionCall) => {c.questions.push(structuredClone(c.questions[0]!));}, - (c: NativePlanQuestionCall) => {c.questions[0]!.options.push(structuredClone(c.questions[0]!.options[0]!));}, - (c: NativePlanQuestionCall) => {c.answers = {[c.questions[0]!.question]: 'unoffered scope decision'};}, - ]) {const c = fresh(); mutate(c); expect(setup(c)).toBe(false);} - expect(engSetupAUQ({...fp(fresh()), signature: 'foreign:call'})).toBe(false); - expect(engSetupAUQ({...fp(fresh()), nativeCall: undefined})).toBe(false); - expect(engSetupAUQ({...fp(fresh()), options: []})).toBe(false); - }); -}); diff --git a/test/eng-task-pause-navigation-f359.test.ts b/test/eng-task-pause-navigation-f359.test.ts deleted file mode 100644 index 4f615eb48..000000000 --- a/test/eng-task-pause-navigation-f359.test.ts +++ /dev/null @@ -1,34 +0,0 @@ -import { test, expect } from 'bun:test'; -import fs from 'node:fs'; -import os from 'node:os'; -import path from 'node:path'; -import { createHash } from 'node:crypto'; -import capture from './fixtures/eng-task-pause-navigation-f359.json'; -import { nativePlanCallFingerprint, hasNativePlanTerminal } from './helpers/claude-pty-runner'; -import type { NativePlanQuestionCall, PlanCountTranscript } from './helpers/plan-count-transcript'; -const actual=()=>({call:structuredClone(capture.transcript.calls.at(-1)!) as NativePlanQuestionCall,prior:structuredClone(capture.transcript.calls.slice(0,-1)) as NativePlanQuestionCall[],plan:capture.plan}); -test('handoff alone never supplies a native terminal or refreshes modifying answers',()=>{ - const x=actual(),fp=nativePlanCallFingerprint(x.call,0,false),admin=new Set([fp.signature]); - const dir=fs.mkdtempSync(path.join(os.tmpdir(),'gstack-eng-task-pause-')),file=path.join(dir,'reviewed.md'),now=Date.now; - try { - fs.writeFileSync(file,x.plan);fs.utimesSync(file,capture.reportSource.mtimeMs/1000,capture.reportSource.mtimeMs/1000); - Date.now=()=>Date.parse('2026-09-16T07:04:00.000Z'); - const t=structuredClone(capture.transcript) as PlanCountTranscript; - const check=(v=t,a=admin)=>hasNativePlanTerminal(v,file,Date.parse('2026-09-16T06:40:00.000Z'),'plan_ready',a); - expect(check()).toBe(false); // Actual capture precedes the native exit. - // The later retained native exit is real; this is a gate replay, not a - // replacement verdict for the original paid timeout/failure. - expect(createHash('sha256').update(JSON.stringify(t.calls)).digest('hex')).toBe(capture.terminalCapture.callsSha256); - t.planReadyRequests=structuredClone(capture.terminalCapture.planReadyRequests); - t.assistantMessages=structuredClone(capture.terminalCapture.assistantMessages); - expect(check()).toBe(true);expect(check(t,new Set())).toBe(false);expect(check(t,new Set(['foreign:call']))).toBe(false); - for(const mutate of [ - (v:PlanCountTranscript)=>{v.planReadyRequests![0]!.failed=true;}, - (v:PlanCountTranscript)=>{v.planReadyRequests![0]!.sessionId='foreign';}, - (v:PlanCountTranscript)=>{v.planReadyRequests![0]!.timestamp=x.call.answeredAt!;}, - (v:PlanCountTranscript)=>{v.planReadyRequests=[];}, - (v:PlanCountTranscript)=>{v.calls[5]!.answeredAt=x.call.answeredAt;}, - (v:PlanCountTranscript)=>{v.calls[5]!.answered=false;v.calls[5]!.unansweredQuestionIndices=[0];}, - ]){const v=structuredClone(t);mutate(v);expect(check(v)).toBe(false);} - }finally{Date.now=now;fs.rmSync(dir,{recursive:true,force:true});} -}); diff --git a/test/helpers/touchfiles-data.ts b/test/helpers/touchfiles-data.ts index 06186a946..ba63cfe71 100644 --- a/test/helpers/touchfiles-data.ts +++ b/test/helpers/touchfiles-data.ts @@ -35,7 +35,7 @@ export const E2E_TOUCHFILES: Record = { 'shared-libs-review-revalidation': ['deslop-shared-libs/**', 'scripts/resolvers/shared-libs.ts', 'scripts/resolvers/index.ts', 'test/helpers/shared-libs-eval-fixture.ts', 'review/**', 'scripts/resolvers/review.ts', 'scripts/resolvers/review-army.ts', 'lib/review-evidence.ts', 'bin/gstack-review-log', 'bin/gstack-review-read', 'bin/gstack-wtree', 'test/skill-e2e-shared-libs.test.ts', 'test/shared-libs-fixture.test.ts', 'test/helpers/e2e-gate.ts', 'scripts/gen-skill-docs.ts', 'test/helpers/agent-sdk-runner.ts', 'lib/claude-bin.ts', 'lib/eval-model.ts', 'test/helpers/shared-libs-review-start-evidence.ts', 'test/shared-libs-review-start-evidence.test.ts', 'test/fixtures/shared-libs-review-start-public.json', 'test/shared-libs-revalidation-prompt.test.ts', 'test/fixtures/shared-libs-revalidation-max-turns-public.json', 'test/fixtures/shared-libs-index-flags-*.json'], 'shared-libs-opportunity-judgment': ['deslop-shared-libs/**', 'scripts/resolvers/shared-libs.ts', 'scripts/resolvers/index.ts', 'test/helpers/shared-libs-eval-fixture.ts', 'test/skill-e2e-shared-libs-periodic.test.ts', 'test/shared-libs-fixture.test.ts', 'test/helpers/e2e-gate.ts', 'scripts/gen-skill-docs.ts', 'test/helpers/llm-judge.ts', 'lib/claude-bin.ts', 'lib/eval-model.ts', 'test/fixtures/shared-libs-readonly-substitution-ci16358.json'], 'shared-libs-pr-coverage': ['deslop-shared-libs/**', 'scripts/resolvers/shared-libs.ts', 'scripts/resolvers/index.ts', 'test/helpers/shared-libs-eval-fixture.ts', 'test/skill-e2e-shared-libs-periodic.test.ts', 'test/shared-libs-fixture.test.ts', 'test/helpers/e2e-gate.ts', 'scripts/gen-skill-docs.ts', 'test/helpers/llm-judge.ts', 'lib/claude-bin.ts', 'lib/eval-model.ts', 'test/fixtures/shared-libs-readonly-substitution-ci16358.json'], - 'shared-libs-plan-callers': ['test/helpers/shared-libs-plan-actor.ts', 'test/shared-libs-plan-actor.test.ts', 'scripts/resolvers/confidence.ts', 'test/helpers/shared-libs-plan-excerpt.ts', 'test/shared-libs-rendering.test.ts', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'deslop-shared-libs/**', 'scripts/resolvers/shared-libs.ts', 'scripts/resolvers/index.ts', 'test/helpers/shared-libs-eval-fixture.ts', 'plan-eng-review/**', 'test/skill-e2e-shared-libs-periodic.test.ts', 'test/eng-scope-entry-ap.test.ts', 'test/plan-scope-recovery-av.test.ts', 'test/fixtures/plan-scope-recovery-av.json', 'test/review-entry-and-design-clarity-au.test.ts', 'scripts/resolvers/preamble/generate-preamble-bash.ts', 'scripts/resolvers/preamble/generate-completion-status.ts', 'test/shared-libs-fixture.test.ts', 'test/helpers/e2e-gate.ts', 'scripts/gen-skill-docs.ts', 'test/helpers/agent-sdk-runner.ts', 'test/helpers/llm-judge.ts', 'lib/claude-bin.ts', 'lib/eval-model.ts'], + 'shared-libs-plan-callers': ['test/helpers/shared-libs-plan-actor.ts', 'test/shared-libs-plan-actor.test.ts', 'scripts/resolvers/confidence.ts', 'test/helpers/shared-libs-plan-excerpt.ts', 'test/shared-libs-rendering.test.ts', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'deslop-shared-libs/**', 'scripts/resolvers/shared-libs.ts', 'scripts/resolvers/index.ts', 'test/helpers/shared-libs-eval-fixture.ts', 'plan-eng-review/**', 'test/skill-e2e-shared-libs-periodic.test.ts', 'test/eng-scope-entry-ap.test.ts', 'test/plan-scope-selection.test.ts', 'test/fixtures/plan-scope-recovery-av.json', 'test/review-entry-and-design-clarity-au.test.ts', 'scripts/resolvers/preamble/generate-preamble-bash.ts', 'scripts/resolvers/preamble/generate-completion-status.ts', 'test/shared-libs-fixture.test.ts', 'test/helpers/e2e-gate.ts', 'scripts/gen-skill-docs.ts', 'test/helpers/agent-sdk-runner.ts', 'test/helpers/llm-judge.ts', 'lib/claude-bin.ts', 'lib/eval-model.ts'], // Browse core (+ test-server dependency) 'browse-basic': ['test/session-runner-stream-lifecycle.test.ts', 'browse/src/**', 'browse/test/test-server.ts', 'test/skill-e2e-bws.test.ts'], 'browse-snapshot': ['test/session-runner-stream-lifecycle.test.ts', 'browse/src/**', 'browse/test/test-server.ts', 'test/skill-e2e-bws.test.ts'], @@ -145,7 +145,7 @@ export const E2E_TOUCHFILES: Record = { 'plan-eng-review': ['test/session-runner-stream-lifecycle.test.ts', 'test/paid-retry-supervision.test.ts', 'test/eng-review-routing.test.ts', 'scripts/resolvers/learnings.ts', - "test/plan-scope-recovery-av.test.ts", + "test/plan-scope-selection.test.ts", "test/fixtures/plan-scope-recovery-av.json", 'test/eng-scope-entry-ap.test.ts', 'plan-eng-review/**', 'test/skill-e2e-plan.test.ts', "test/review-entry-and-design-clarity-au.test.ts", "scripts/resolvers/preamble/generate-preamble-bash.ts", "scripts/resolvers/preamble/generate-completion-status.ts", 'scripts/resolvers/testing.ts', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/review.ts', 'test/plan-review-cases.test.ts' @@ -153,7 +153,7 @@ export const E2E_TOUCHFILES: Record = { 'plan-eng-review-artifact': ['test/session-runner-stream-lifecycle.test.ts', 'test/paid-retry-supervision.test.ts', 'test/eng-review-routing.test.ts', 'scripts/resolvers/learnings.ts', - "test/plan-scope-recovery-av.test.ts", + "test/plan-scope-selection.test.ts", "test/fixtures/plan-scope-recovery-av.json", 'test/eng-scope-entry-ap.test.ts', 'plan-eng-review/**', 'test/skill-e2e-plan.test.ts', "test/review-entry-and-design-clarity-au.test.ts", "scripts/resolvers/preamble/generate-preamble-bash.ts", "scripts/resolvers/preamble/generate-completion-status.ts", 'scripts/resolvers/testing.ts', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/review.ts', 'test/plan-review-cases.test.ts' @@ -163,7 +163,7 @@ export const E2E_TOUCHFILES: Record = { 'test/helpers/office-hours-attempt.ts', 'test/office-hours-attempt.test.ts', 'test/plan-review-report-recording.test.ts', 'test/fixtures/plan-review-report-public.json', 'scripts/resolvers/learnings.ts', - "test/plan-scope-recovery-av.test.ts", + "test/plan-scope-selection.test.ts", "test/fixtures/plan-scope-recovery-av.json", 'test/eng-scope-entry-ap.test.ts', 'plan-eng-review/**', 'scripts/gen-skill-docs.ts', 'test/skill-e2e-plan.test.ts', "test/review-entry-and-design-clarity-au.test.ts", "scripts/resolvers/preamble/generate-preamble-bash.ts", "scripts/resolvers/preamble/generate-completion-status.ts", 'scripts/resolvers/testing.ts', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/review.ts', 'test/plan-review-cases.test.ts' @@ -180,26 +180,26 @@ export const E2E_TOUCHFILES: Record = { 'test/ceo-plan-mode-fixture.test.ts', 'test/helpers/plan-count-fixture.ts', 'test/plan-count-fixture.test.ts', - 'test/auto-decide-recommendation-scope.test.ts', + 'test/fixtures/auto-decide-recommendation-361c.json', - 'test/auto-decide-target-identity.test.ts', + 'test/fixtures/auto-decide-target-361c.json', "test/fixtures/plan-scope-target-aw.json", 'test/pty-screen-unicode-ap.test.ts', - 'test/auto-decide-saved-ai.test.ts', 'test/fixtures/auto-decide-saved-ai.json', 'test/fixtures/auto-decide-retry-ai.json','bin/gstack-skill-start', 'bin/gstack-skill-end', 'plan-ceo-review/**', 'scripts/resolvers/preamble/generate-completion-status.ts', 'scripts/resolvers/question-tuning.ts', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/preamble.ts', 'scripts/resolvers/review.ts', 'test/helpers/claude-pty-runner.ts', 'test/plan-count-design-ui-recovery.test.ts', 'test/fixtures/design-ui-boxed-question.json', 'test/helpers/hermetic-skill-runtime.ts', 'test/hermetic-skill-runtime.test.ts', 'test/helpers/pty-trust-dialog.ts', 'test/pty-trust-dialog.test.ts', 'test/skill-e2e-plan-ceo-plan-mode.test.ts', 'test/plan-count-truncated-border.test.ts', 'test/fixtures/eng-d2-truncated-border-0bcd.json', 'test/plan-count-truncated-question.test.ts', 'test/plan-count-clipped-elision.test.ts', 'test/fixtures/eng-d1-clipped-elision-1579.json', 'test/fixtures/eng-d2-planning-prelude-4d.json', 'test/fixtures/ceo-approach-z-call.json', 'test/fixtures/ceo-approach-z-screen.txt', 'test/helpers/plan-scope-selection.ts', 'test/plan-scope-selection.test.ts', 'test/helpers/native-auto-decide.ts', 'test/auto-decide-current-declaration.test.ts', 'test/fixtures/auto-decide-current-declaration-6aef.json', 'test/auto-decide-explanatory-mode.test.ts', 'test/fixtures/auto-decide-explanatory-mode-043a.json', 'test/fixtures/auto-decide-explanatory-mode-749df.json', 'test/auto-decide-structured.test.ts', 'test/fixtures/auto-decide-structured-77.json', 'test/helpers/auto-decision-state.ts', 'test/auto-decision-state.test.ts', 'test/fixtures/auto-decide-state-cab3.json', 'bin/gstack-question-log', 'bin/gstack-question-preference', 'test/native-auto-decide.test.ts', 'test/native-auto-decide-pty.test.ts', 'test/helpers/fake-plan-seed.ts', 'test/helpers/plan-seed-submission.ts', 'test/plan-seed-submission.test.ts', 'test/fixtures/plan-seed-cli.ts', 'test/fixtures/native-auto-decide-ag.json', 'test/eng-seeded-completion-ai.test.ts', 'test/fixtures/eng-seeded-completion-ai.json', 'test/helpers/plan-count-pending-exit.ts', 'test/plan-count-pending-exit.test.ts', 'test/helpers/pty-screen.ts', 'test/pty-screen.test.ts', 'test/pty-screen-session.test.ts', 'test/fixtures/pty-screen/**', - 'test/design-scope-declaration-ak.test.ts', - 'test/design-scope-announcement-ao.test.ts', 'test/fixtures/design-scope-announcement-ao.json', +'test/fixtures/auto-decide-saved-ai.json', 'test/fixtures/auto-decide-retry-ai.json','bin/gstack-skill-start', 'bin/gstack-skill-end', 'plan-ceo-review/**', 'scripts/resolvers/preamble/generate-completion-status.ts', 'scripts/resolvers/question-tuning.ts', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/preamble.ts', 'scripts/resolvers/review.ts', 'test/helpers/claude-pty-runner.ts', 'test/plan-count-design-ui-recovery.test.ts', 'test/fixtures/design-ui-boxed-question.json', 'test/helpers/hermetic-skill-runtime.ts', 'test/hermetic-skill-runtime.test.ts', 'test/helpers/pty-trust-dialog.ts', 'test/pty-trust-dialog.test.ts', 'test/skill-e2e-plan-ceo-plan-mode.test.ts', 'test/plan-count-truncated-border.test.ts', 'test/fixtures/eng-d2-truncated-border-0bcd.json', 'test/plan-count-truncated-question.test.ts', 'test/plan-count-clipped-elision.test.ts', 'test/fixtures/eng-d1-clipped-elision-1579.json', 'test/fixtures/eng-d2-planning-prelude-4d.json', 'test/fixtures/ceo-approach-z-call.json', 'test/fixtures/ceo-approach-z-screen.txt', 'test/helpers/plan-scope-selection.ts', 'test/plan-scope-selection.test.ts', 'test/helpers/native-auto-decide.ts','test/fixtures/auto-decide-current-declaration-6aef.json','test/fixtures/auto-decide-explanatory-mode-043a.json', 'test/fixtures/auto-decide-explanatory-mode-749df.json','test/fixtures/auto-decide-structured-77.json', 'test/helpers/auto-decision-state.ts','test/fixtures/auto-decide-state-cab3.json', 'bin/gstack-question-log', 'bin/gstack-question-preference', 'test/native-auto-decide.test.ts', 'test/native-auto-decide-pty.test.ts', 'test/helpers/fake-plan-seed.ts', 'test/helpers/plan-seed-submission.ts', 'test/plan-seed-submission.test.ts', 'test/fixtures/plan-seed-cli.ts', 'test/fixtures/native-auto-decide-ag.json', 'test/eng-seeded-completion-ai.test.ts', 'test/fixtures/eng-seeded-completion-ai.json', 'test/helpers/plan-count-pending-exit.ts', 'test/plan-count-pending-exit.test.ts', 'test/helpers/pty-screen.ts', 'test/pty-screen.test.ts', 'test/pty-screen-session.test.ts', 'test/fixtures/pty-screen/**', + +'test/fixtures/design-scope-announcement-ao.json', 'test/fixtures/design-scope-declaration-ak.json', - "test/eng-option-b-scope-al.test.ts", + "test/fixtures/eng-option-b-scope-al.json", 'test/pty-workspace-trust.test.ts', 'test/fixtures/pty-companion-cli.ts', 'lib/fs-atomic.ts', 'test/helpers/plan-skill-questions.ts', 'test/fixtures/eng-auq-validation-error.json', 'test/fixtures/bash-directory-permission.json', 'test/fixtures/design-tasks-bash-permission.json', 'test/plan-skill-read-permission.test.ts', 'test/fixtures/read-permission.json', 'test/pty-numbered-option-indent-native.test.ts', 'test/fixtures/ceo-split-e5-numbered-description-491.json', 'test/plan-skill-questions.test.ts', 'test/helpers/plan-skill-question-events.ts', 'test/plan-skill-question-events.test.ts', 'test/helpers/plan-skill-question-hook-scope.ts', 'test/helpers/skill-census.ts', 'test/plan-skill-question-hook-scope.test.ts', 'scripts/resolvers/tasks-section.ts' ], 'plan-eng-review-plan-mode': [ 'lib/claude-public-transcript.ts', - 'test/auto-decide-recommendation-scope.test.ts', + 'test/fixtures/auto-decide-recommendation-361c.json', - 'test/auto-decide-target-identity.test.ts', + 'test/fixtures/auto-decide-target-361c.json', 'test/paid-retry-supervision.test.ts', 'test/autoplan-public-narration.test.ts', 'test/fixtures/autoplan-public-narration-ad.json', @@ -207,13 +207,13 @@ export const E2E_TOUCHFILES: Record = { 'scripts/resolvers/learnings.ts', "test/fixtures/plan-scope-target-aw.json", - "test/plan-scope-recovery-av.test.ts", + "test/fixtures/plan-scope-recovery-av.json", 'test/pty-screen-unicode-ap.test.ts', 'test/eng-scope-entry-ap.test.ts', - 'test/auto-decide-saved-ai.test.ts', 'test/fixtures/auto-decide-saved-ai.json', 'test/fixtures/auto-decide-retry-ai.json','bin/gstack-skill-start', 'bin/gstack-skill-end', 'plan-eng-review/**', 'scripts/resolvers/preamble/generate-completion-status.ts', 'scripts/resolvers/question-tuning.ts', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/preamble.ts', 'scripts/resolvers/review.ts', 'test/helpers/claude-pty-runner.ts', 'test/plan-count-design-ui-recovery.test.ts', 'test/fixtures/design-ui-boxed-question.json', 'test/helpers/hermetic-skill-runtime.ts', 'test/hermetic-skill-runtime.test.ts', 'test/helpers/pty-trust-dialog.ts', 'test/pty-trust-dialog.test.ts', 'test/skill-e2e-plan-eng-plan-mode.test.ts', 'test/plan-count-truncated-border.test.ts', 'test/fixtures/eng-d2-truncated-border-0bcd.json', 'test/plan-count-truncated-question.test.ts', 'test/plan-count-clipped-elision.test.ts', 'test/fixtures/eng-d1-clipped-elision-1579.json', 'test/fixtures/eng-d2-planning-prelude-4d.json', 'test/fixtures/ceo-approach-z-call.json', 'test/fixtures/ceo-approach-z-screen.txt', 'test/helpers/plan-scope-selection.ts', 'test/plan-scope-selection.test.ts', 'test/fixtures/design-plan-scope-ag.json', 'test/design-scope-selection-aj.test.ts', 'test/fixtures/design-scope-selection-aj.json', 'test/helpers/native-auto-decide.ts', 'test/auto-decide-current-declaration.test.ts', 'test/fixtures/auto-decide-current-declaration-6aef.json', 'test/auto-decide-explanatory-mode.test.ts', 'test/fixtures/auto-decide-explanatory-mode-043a.json', 'test/fixtures/auto-decide-explanatory-mode-749df.json', 'test/auto-decide-structured.test.ts', 'test/fixtures/auto-decide-structured-77.json', 'test/helpers/auto-decision-state.ts', 'test/auto-decision-state.test.ts', 'test/fixtures/auto-decide-state-cab3.json', 'bin/gstack-question-log', 'bin/gstack-question-preference', 'test/native-auto-decide.test.ts', 'test/native-auto-decide-pty.test.ts', 'test/helpers/fake-plan-seed.ts', 'test/fixtures/native-auto-decide-ag.json', 'test/eng-seeded-completion-ai.test.ts', 'test/fixtures/eng-seeded-completion-ai.json', 'test/helpers/plan-count-pending-exit.ts', 'test/plan-count-pending-exit.test.ts', 'test/helpers/pty-screen.ts', 'test/pty-screen.test.ts', 'test/pty-screen-session.test.ts', 'test/fixtures/pty-screen/**', - 'test/design-scope-declaration-ak.test.ts', - 'test/design-scope-announcement-ao.test.ts', 'test/fixtures/design-scope-announcement-ao.json', +'test/fixtures/auto-decide-saved-ai.json', 'test/fixtures/auto-decide-retry-ai.json','bin/gstack-skill-start', 'bin/gstack-skill-end', 'plan-eng-review/**', 'scripts/resolvers/preamble/generate-completion-status.ts', 'scripts/resolvers/question-tuning.ts', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/preamble.ts', 'scripts/resolvers/review.ts', 'test/helpers/claude-pty-runner.ts', 'test/plan-count-design-ui-recovery.test.ts', 'test/fixtures/design-ui-boxed-question.json', 'test/helpers/hermetic-skill-runtime.ts', 'test/hermetic-skill-runtime.test.ts', 'test/helpers/pty-trust-dialog.ts', 'test/pty-trust-dialog.test.ts', 'test/skill-e2e-plan-eng-plan-mode.test.ts', 'test/plan-count-truncated-border.test.ts', 'test/fixtures/eng-d2-truncated-border-0bcd.json', 'test/plan-count-truncated-question.test.ts', 'test/plan-count-clipped-elision.test.ts', 'test/fixtures/eng-d1-clipped-elision-1579.json', 'test/fixtures/eng-d2-planning-prelude-4d.json', 'test/fixtures/ceo-approach-z-call.json', 'test/fixtures/ceo-approach-z-screen.txt', 'test/helpers/plan-scope-selection.ts', 'test/plan-scope-selection.test.ts', 'test/fixtures/design-plan-scope-ag.json','test/fixtures/design-scope-selection-aj.json', 'test/helpers/native-auto-decide.ts','test/fixtures/auto-decide-current-declaration-6aef.json','test/fixtures/auto-decide-explanatory-mode-043a.json', 'test/fixtures/auto-decide-explanatory-mode-749df.json','test/fixtures/auto-decide-structured-77.json', 'test/helpers/auto-decision-state.ts','test/fixtures/auto-decide-state-cab3.json', 'bin/gstack-question-log', 'bin/gstack-question-preference', 'test/native-auto-decide.test.ts', 'test/native-auto-decide-pty.test.ts', 'test/helpers/fake-plan-seed.ts', 'test/fixtures/native-auto-decide-ag.json', 'test/eng-seeded-completion-ai.test.ts', 'test/fixtures/eng-seeded-completion-ai.json', 'test/helpers/plan-count-pending-exit.ts', 'test/plan-count-pending-exit.test.ts', 'test/helpers/pty-screen.ts', 'test/pty-screen.test.ts', 'test/pty-screen-session.test.ts', 'test/fixtures/pty-screen/**', + +'test/fixtures/design-scope-announcement-ao.json', 'test/fixtures/design-scope-declaration-ak.json', - "test/eng-option-b-scope-al.test.ts", + "test/fixtures/eng-option-b-scope-al.json", "test/review-entry-and-design-clarity-au.test.ts", "scripts/resolvers/preamble/generate-preamble-bash.ts", @@ -221,9 +221,9 @@ export const E2E_TOUCHFILES: Record = { ], 'plan-design-review-plan-mode': ['test/session-runner-stream-lifecycle.test.ts', 'lib/claude-public-transcript.ts', - 'test/auto-decide-recommendation-scope.test.ts', + 'test/fixtures/auto-decide-recommendation-361c.json', - 'test/auto-decide-target-identity.test.ts', + 'test/fixtures/auto-decide-target-361c.json', 'test/autoplan-public-narration.test.ts', 'test/fixtures/autoplan-public-narration-ad.json', @@ -232,33 +232,33 @@ export const E2E_TOUCHFILES: Record = { 'test/plan-design-sdk-fixture.test.ts', "test/fixtures/plan-scope-target-aw.json", - "test/plan-scope-recovery-av.test.ts", + "test/fixtures/plan-scope-recovery-av.json", "test/fixtures/design-scope-checkpoint-at.json",'test/pty-screen-unicode-ap.test.ts', - 'test/auto-decide-saved-ai.test.ts', 'test/fixtures/auto-decide-saved-ai.json', 'test/fixtures/auto-decide-retry-ai.json','bin/gstack-skill-start', 'bin/gstack-skill-end', 'plan-design-review/**', 'scripts/resolvers/preamble/generate-completion-status.ts', 'scripts/resolvers/question-tuning.ts', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/preamble.ts', 'scripts/resolvers/review.ts', 'test/helpers/claude-pty-runner.ts', 'test/plan-count-design-ui-recovery.test.ts', 'test/fixtures/design-ui-boxed-question.json', 'test/helpers/hermetic-skill-runtime.ts', 'test/hermetic-skill-runtime.test.ts', 'test/helpers/pty-trust-dialog.ts', 'test/pty-trust-dialog.test.ts', 'test/skill-e2e-plan-design-plan-mode.test.ts', 'test/skill-e2e-design.test.ts', 'test/plan-count-truncated-border.test.ts', 'test/fixtures/eng-d2-truncated-border-0bcd.json', 'test/plan-count-truncated-question.test.ts', 'test/plan-count-clipped-elision.test.ts', 'test/fixtures/eng-d1-clipped-elision-1579.json', 'test/fixtures/eng-d2-planning-prelude-4d.json', 'test/fixtures/ceo-approach-z-call.json', 'test/fixtures/ceo-approach-z-screen.txt', 'test/helpers/plan-scope-selection.ts', 'test/plan-scope-selection.test.ts', 'test/fixtures/design-plan-scope-ag.json', 'test/design-scope-selection-aj.test.ts', 'test/fixtures/design-scope-selection-aj.json', 'test/helpers/native-auto-decide.ts', 'test/auto-decide-current-declaration.test.ts', 'test/fixtures/auto-decide-current-declaration-6aef.json', 'test/auto-decide-explanatory-mode.test.ts', 'test/fixtures/auto-decide-explanatory-mode-043a.json', 'test/fixtures/auto-decide-explanatory-mode-749df.json', 'test/auto-decide-structured.test.ts', 'test/fixtures/auto-decide-structured-77.json', 'test/helpers/auto-decision-state.ts', 'test/auto-decision-state.test.ts', 'test/fixtures/auto-decide-state-cab3.json', 'bin/gstack-question-log', 'bin/gstack-question-preference', 'test/native-auto-decide.test.ts', 'test/native-auto-decide-pty.test.ts', 'test/helpers/fake-plan-seed.ts', 'test/fixtures/native-auto-decide-ag.json', 'test/eng-seeded-completion-ai.test.ts', 'test/fixtures/eng-seeded-completion-ai.json', 'test/helpers/plan-count-pending-exit.ts', 'test/plan-count-pending-exit.test.ts', 'test/helpers/pty-screen.ts', 'test/pty-screen.test.ts', 'test/pty-screen-session.test.ts', 'test/fixtures/pty-screen/**', - 'test/design-scope-declaration-ak.test.ts', - 'test/design-scope-announcement-ao.test.ts', 'test/fixtures/design-scope-announcement-ao.json', +'test/fixtures/auto-decide-saved-ai.json', 'test/fixtures/auto-decide-retry-ai.json','bin/gstack-skill-start', 'bin/gstack-skill-end', 'plan-design-review/**', 'scripts/resolvers/preamble/generate-completion-status.ts', 'scripts/resolvers/question-tuning.ts', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/preamble.ts', 'scripts/resolvers/review.ts', 'test/helpers/claude-pty-runner.ts', 'test/plan-count-design-ui-recovery.test.ts', 'test/fixtures/design-ui-boxed-question.json', 'test/helpers/hermetic-skill-runtime.ts', 'test/hermetic-skill-runtime.test.ts', 'test/helpers/pty-trust-dialog.ts', 'test/pty-trust-dialog.test.ts', 'test/skill-e2e-plan-design-plan-mode.test.ts', 'test/skill-e2e-design.test.ts', 'test/plan-count-truncated-border.test.ts', 'test/fixtures/eng-d2-truncated-border-0bcd.json', 'test/plan-count-truncated-question.test.ts', 'test/plan-count-clipped-elision.test.ts', 'test/fixtures/eng-d1-clipped-elision-1579.json', 'test/fixtures/eng-d2-planning-prelude-4d.json', 'test/fixtures/ceo-approach-z-call.json', 'test/fixtures/ceo-approach-z-screen.txt', 'test/helpers/plan-scope-selection.ts', 'test/plan-scope-selection.test.ts', 'test/fixtures/design-plan-scope-ag.json','test/fixtures/design-scope-selection-aj.json', 'test/helpers/native-auto-decide.ts','test/fixtures/auto-decide-current-declaration-6aef.json','test/fixtures/auto-decide-explanatory-mode-043a.json', 'test/fixtures/auto-decide-explanatory-mode-749df.json','test/fixtures/auto-decide-structured-77.json', 'test/helpers/auto-decision-state.ts','test/fixtures/auto-decide-state-cab3.json', 'bin/gstack-question-log', 'bin/gstack-question-preference', 'test/native-auto-decide.test.ts', 'test/native-auto-decide-pty.test.ts', 'test/helpers/fake-plan-seed.ts', 'test/fixtures/native-auto-decide-ag.json', 'test/eng-seeded-completion-ai.test.ts', 'test/fixtures/eng-seeded-completion-ai.json', 'test/helpers/plan-count-pending-exit.ts', 'test/plan-count-pending-exit.test.ts', 'test/helpers/pty-screen.ts', 'test/pty-screen.test.ts', 'test/pty-screen-session.test.ts', 'test/fixtures/pty-screen/**', + +'test/fixtures/design-scope-announcement-ao.json', 'test/fixtures/design-scope-declaration-ak.json', - "test/eng-option-b-scope-al.test.ts", + "test/fixtures/eng-option-b-scope-al.json", - "test/design-scope-entry-aq.test.ts", + "test/review-entry-and-design-clarity-au.test.ts", "scripts/resolvers/preamble/generate-preamble-bash.ts", 'test/pty-workspace-trust.test.ts', 'test/fixtures/pty-companion-cli.ts', 'test/helpers/plan-seed-submission.ts', 'test/plan-seed-submission.test.ts', 'test/fixtures/plan-seed-cli.ts', 'test/helpers/owned-claude-transcript.ts', 'lib/fs-atomic.ts', 'test/helpers/plan-skill-questions.ts', 'test/fixtures/eng-auq-validation-error.json', 'test/fixtures/bash-directory-permission.json', 'test/fixtures/design-tasks-bash-permission.json', 'test/plan-skill-read-permission.test.ts', 'test/fixtures/read-permission.json', 'test/pty-numbered-option-indent-native.test.ts', 'test/fixtures/ceo-split-e5-numbered-description-491.json', 'test/plan-skill-questions.test.ts', 'test/helpers/plan-skill-question-events.ts', 'test/plan-skill-question-events.test.ts', 'test/helpers/plan-skill-question-hook-scope.ts', 'test/helpers/skill-census.ts', 'test/plan-skill-question-hook-scope.test.ts', 'test/helpers/plan-mode-evidence.ts', 'test/plan-mode-evidence.test.ts', 'lib/redact-engine.ts', 'lib/redact-patterns.ts' ], 'plan-devex-review-plan-mode': [ - 'test/auto-decide-recommendation-scope.test.ts', + 'test/fixtures/auto-decide-recommendation-361c.json', - 'test/auto-decide-target-identity.test.ts', + 'test/fixtures/auto-decide-target-361c.json', "test/fixtures/plan-scope-target-aw.json", 'test/pty-screen-unicode-ap.test.ts', - 'test/auto-decide-saved-ai.test.ts', 'test/fixtures/auto-decide-saved-ai.json', 'test/fixtures/auto-decide-retry-ai.json','bin/gstack-skill-start', 'bin/gstack-skill-end', 'plan-devex-review/**', 'scripts/resolvers/preamble/generate-completion-status.ts', 'scripts/resolvers/question-tuning.ts', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/preamble.ts', 'scripts/resolvers/review.ts', 'test/helpers/claude-pty-runner.ts', 'test/plan-count-design-ui-recovery.test.ts', 'test/fixtures/design-ui-boxed-question.json', 'test/helpers/hermetic-skill-runtime.ts', 'test/hermetic-skill-runtime.test.ts', 'test/helpers/pty-trust-dialog.ts', 'test/pty-trust-dialog.test.ts', 'test/skill-e2e-plan-devex-plan-mode.test.ts', 'test/plan-count-truncated-border.test.ts', 'test/fixtures/eng-d2-truncated-border-0bcd.json', 'test/plan-count-truncated-question.test.ts', 'test/plan-count-clipped-elision.test.ts', 'test/fixtures/eng-d1-clipped-elision-1579.json', 'test/fixtures/eng-d2-planning-prelude-4d.json', 'test/fixtures/ceo-approach-z-call.json', 'test/fixtures/ceo-approach-z-screen.txt', 'test/helpers/plan-scope-selection.ts', 'test/plan-scope-selection.test.ts', 'test/helpers/native-auto-decide.ts', 'test/auto-decide-current-declaration.test.ts', 'test/fixtures/auto-decide-current-declaration-6aef.json', 'test/auto-decide-explanatory-mode.test.ts', 'test/fixtures/auto-decide-explanatory-mode-043a.json', 'test/fixtures/auto-decide-explanatory-mode-749df.json', 'test/auto-decide-structured.test.ts', 'test/fixtures/auto-decide-structured-77.json', 'test/helpers/auto-decision-state.ts', 'test/auto-decision-state.test.ts', 'test/fixtures/auto-decide-state-cab3.json', 'bin/gstack-question-log', 'bin/gstack-question-preference', 'test/native-auto-decide.test.ts', 'test/native-auto-decide-pty.test.ts', 'test/helpers/fake-plan-seed.ts', 'test/helpers/plan-seed-submission.ts', 'test/plan-seed-submission.test.ts', 'test/fixtures/plan-seed-cli.ts', 'test/fixtures/native-auto-decide-ag.json', 'test/eng-seeded-completion-ai.test.ts', 'test/fixtures/eng-seeded-completion-ai.json', 'test/helpers/plan-count-pending-exit.ts', 'test/plan-count-pending-exit.test.ts', 'test/helpers/pty-screen.ts', 'test/pty-screen.test.ts', 'test/pty-screen-session.test.ts', 'test/fixtures/pty-screen/**', - 'test/design-scope-declaration-ak.test.ts', - 'test/design-scope-announcement-ao.test.ts', 'test/fixtures/design-scope-announcement-ao.json', +'test/fixtures/auto-decide-saved-ai.json', 'test/fixtures/auto-decide-retry-ai.json','bin/gstack-skill-start', 'bin/gstack-skill-end', 'plan-devex-review/**', 'scripts/resolvers/preamble/generate-completion-status.ts', 'scripts/resolvers/question-tuning.ts', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/preamble.ts', 'scripts/resolvers/review.ts', 'test/helpers/claude-pty-runner.ts', 'test/plan-count-design-ui-recovery.test.ts', 'test/fixtures/design-ui-boxed-question.json', 'test/helpers/hermetic-skill-runtime.ts', 'test/hermetic-skill-runtime.test.ts', 'test/helpers/pty-trust-dialog.ts', 'test/pty-trust-dialog.test.ts', 'test/skill-e2e-plan-devex-plan-mode.test.ts', 'test/plan-count-truncated-border.test.ts', 'test/fixtures/eng-d2-truncated-border-0bcd.json', 'test/plan-count-truncated-question.test.ts', 'test/plan-count-clipped-elision.test.ts', 'test/fixtures/eng-d1-clipped-elision-1579.json', 'test/fixtures/eng-d2-planning-prelude-4d.json', 'test/fixtures/ceo-approach-z-call.json', 'test/fixtures/ceo-approach-z-screen.txt', 'test/helpers/plan-scope-selection.ts', 'test/plan-scope-selection.test.ts', 'test/helpers/native-auto-decide.ts','test/fixtures/auto-decide-current-declaration-6aef.json','test/fixtures/auto-decide-explanatory-mode-043a.json', 'test/fixtures/auto-decide-explanatory-mode-749df.json','test/fixtures/auto-decide-structured-77.json', 'test/helpers/auto-decision-state.ts','test/fixtures/auto-decide-state-cab3.json', 'bin/gstack-question-log', 'bin/gstack-question-preference', 'test/native-auto-decide.test.ts', 'test/native-auto-decide-pty.test.ts', 'test/helpers/fake-plan-seed.ts', 'test/helpers/plan-seed-submission.ts', 'test/plan-seed-submission.test.ts', 'test/fixtures/plan-seed-cli.ts', 'test/fixtures/native-auto-decide-ag.json', 'test/eng-seeded-completion-ai.test.ts', 'test/fixtures/eng-seeded-completion-ai.json', 'test/helpers/plan-count-pending-exit.ts', 'test/plan-count-pending-exit.test.ts', 'test/helpers/pty-screen.ts', 'test/pty-screen.test.ts', 'test/pty-screen-session.test.ts', 'test/fixtures/pty-screen/**', + +'test/fixtures/design-scope-announcement-ao.json', 'test/fixtures/design-scope-declaration-ak.json', - "test/eng-option-b-scope-al.test.ts", + "test/fixtures/eng-option-b-scope-al.json", 'test/pty-workspace-trust.test.ts', 'test/fixtures/pty-companion-cli.ts', 'lib/fs-atomic.ts', 'test/helpers/plan-skill-questions.ts', 'test/fixtures/eng-auq-validation-error.json', 'test/fixtures/bash-directory-permission.json', 'test/fixtures/design-tasks-bash-permission.json', 'test/plan-skill-read-permission.test.ts', 'test/fixtures/read-permission.json', 'test/pty-numbered-option-indent-native.test.ts', 'test/fixtures/ceo-split-e5-numbered-description-491.json', 'test/plan-skill-questions.test.ts', 'test/helpers/plan-skill-question-events.ts', 'test/plan-skill-question-events.test.ts', 'test/helpers/plan-skill-question-hook-scope.ts', 'test/helpers/skill-census.ts', 'test/plan-skill-question-hook-scope.test.ts' ], @@ -269,24 +269,24 @@ export const E2E_TOUCHFILES: Record = { // pass of each, sharing the API budget with sibling tests — not the // sequential ~+10min a local read suggests. 'plan-mode-no-op': [ - 'test/auto-decide-recommendation-scope.test.ts', + 'test/fixtures/auto-decide-recommendation-361c.json', - 'test/auto-decide-target-identity.test.ts', + 'test/fixtures/auto-decide-target-361c.json', 'test/paid-retry-supervision.test.ts', 'scripts/resolvers/learnings.ts', "test/fixtures/plan-scope-target-aw.json", - "test/plan-scope-recovery-av.test.ts", + "test/fixtures/plan-scope-recovery-av.json", "test/fixtures/design-scope-checkpoint-at.json",'test/pty-screen-unicode-ap.test.ts', 'test/eng-scope-entry-ap.test.ts', - 'test/auto-decide-saved-ai.test.ts', 'test/fixtures/auto-decide-saved-ai.json', 'test/fixtures/auto-decide-retry-ai.json','bin/gstack-skill-start', 'bin/gstack-skill-end', 'plan-ceo-review/**', 'plan-eng-review/**', 'plan-design-review/**', 'scripts/resolvers/preamble/generate-completion-status.ts', 'scripts/resolvers/preamble.ts', 'test/helpers/claude-pty-runner.ts', 'test/plan-count-design-ui-recovery.test.ts', 'test/fixtures/design-ui-boxed-question.json', 'test/helpers/hermetic-skill-runtime.ts', 'test/hermetic-skill-runtime.test.ts', 'test/helpers/pty-trust-dialog.ts', 'test/pty-trust-dialog.test.ts', 'test/skill-e2e-plan-mode-no-op.test.ts', 'test/plan-count-truncated-border.test.ts', 'test/fixtures/eng-d2-truncated-border-0bcd.json', 'test/plan-count-truncated-question.test.ts', 'test/plan-count-clipped-elision.test.ts', 'test/fixtures/eng-d1-clipped-elision-1579.json', 'test/fixtures/eng-d2-planning-prelude-4d.json', 'test/fixtures/ceo-approach-z-call.json', 'test/fixtures/ceo-approach-z-screen.txt', 'test/helpers/plan-scope-selection.ts', 'test/plan-scope-selection.test.ts', 'test/helpers/native-auto-decide.ts', 'test/auto-decide-current-declaration.test.ts', 'test/fixtures/auto-decide-current-declaration-6aef.json', 'test/auto-decide-explanatory-mode.test.ts', 'test/fixtures/auto-decide-explanatory-mode-043a.json', 'test/fixtures/auto-decide-explanatory-mode-749df.json', 'test/auto-decide-structured.test.ts', 'test/fixtures/auto-decide-structured-77.json', 'test/helpers/auto-decision-state.ts', 'test/auto-decision-state.test.ts', 'test/fixtures/auto-decide-state-cab3.json', 'bin/gstack-question-log', 'bin/gstack-question-preference', 'test/native-auto-decide.test.ts', 'test/native-auto-decide-pty.test.ts', 'test/helpers/fake-plan-seed.ts', 'test/fixtures/native-auto-decide-ag.json', 'test/eng-seeded-completion-ai.test.ts', 'test/fixtures/eng-seeded-completion-ai.json', 'test/helpers/plan-count-pending-exit.ts', 'test/plan-count-pending-exit.test.ts', 'test/helpers/pty-screen.ts', 'test/pty-screen.test.ts', 'test/pty-screen-session.test.ts', 'test/fixtures/pty-screen/**', - 'test/design-scope-declaration-ak.test.ts', - 'test/design-scope-announcement-ao.test.ts', 'test/fixtures/design-scope-announcement-ao.json', +'test/fixtures/auto-decide-saved-ai.json', 'test/fixtures/auto-decide-retry-ai.json','bin/gstack-skill-start', 'bin/gstack-skill-end', 'plan-ceo-review/**', 'plan-eng-review/**', 'plan-design-review/**', 'scripts/resolvers/preamble/generate-completion-status.ts', 'scripts/resolvers/preamble.ts', 'test/helpers/claude-pty-runner.ts', 'test/plan-count-design-ui-recovery.test.ts', 'test/fixtures/design-ui-boxed-question.json', 'test/helpers/hermetic-skill-runtime.ts', 'test/hermetic-skill-runtime.test.ts', 'test/helpers/pty-trust-dialog.ts', 'test/pty-trust-dialog.test.ts', 'test/skill-e2e-plan-mode-no-op.test.ts', 'test/plan-count-truncated-border.test.ts', 'test/fixtures/eng-d2-truncated-border-0bcd.json', 'test/plan-count-truncated-question.test.ts', 'test/plan-count-clipped-elision.test.ts', 'test/fixtures/eng-d1-clipped-elision-1579.json', 'test/fixtures/eng-d2-planning-prelude-4d.json', 'test/fixtures/ceo-approach-z-call.json', 'test/fixtures/ceo-approach-z-screen.txt', 'test/helpers/plan-scope-selection.ts', 'test/plan-scope-selection.test.ts', 'test/helpers/native-auto-decide.ts','test/fixtures/auto-decide-current-declaration-6aef.json','test/fixtures/auto-decide-explanatory-mode-043a.json', 'test/fixtures/auto-decide-explanatory-mode-749df.json','test/fixtures/auto-decide-structured-77.json', 'test/helpers/auto-decision-state.ts','test/fixtures/auto-decide-state-cab3.json', 'bin/gstack-question-log', 'bin/gstack-question-preference', 'test/native-auto-decide.test.ts', 'test/native-auto-decide-pty.test.ts', 'test/helpers/fake-plan-seed.ts', 'test/fixtures/native-auto-decide-ag.json', 'test/eng-seeded-completion-ai.test.ts', 'test/fixtures/eng-seeded-completion-ai.json', 'test/helpers/plan-count-pending-exit.ts', 'test/plan-count-pending-exit.test.ts', 'test/helpers/pty-screen.ts', 'test/pty-screen.test.ts', 'test/pty-screen-session.test.ts', 'test/fixtures/pty-screen/**', + +'test/fixtures/design-scope-announcement-ao.json', 'test/fixtures/design-scope-declaration-ak.json', - "test/eng-option-b-scope-al.test.ts", + "test/fixtures/eng-option-b-scope-al.json", - "test/design-scope-entry-aq.test.ts", + "test/review-entry-and-design-clarity-au.test.ts", "scripts/resolvers/preamble/generate-preamble-bash.ts", 'test/pty-workspace-trust.test.ts', 'test/fixtures/pty-companion-cli.ts', 'test/helpers/plan-seed-submission.ts', 'test/plan-seed-submission.test.ts', 'test/fixtures/plan-seed-cli.ts', 'test/helpers/owned-claude-transcript.ts', 'lib/fs-atomic.ts', 'test/helpers/plan-skill-questions.ts', 'test/fixtures/eng-auq-validation-error.json', 'test/fixtures/bash-directory-permission.json', 'test/fixtures/design-tasks-bash-permission.json', 'test/plan-skill-read-permission.test.ts', 'test/fixtures/read-permission.json', 'test/pty-numbered-option-indent-native.test.ts', 'test/fixtures/ceo-split-e5-numbered-description-491.json', 'test/plan-skill-questions.test.ts', 'test/helpers/plan-skill-question-events.ts', 'test/plan-skill-question-events.test.ts', 'test/helpers/plan-skill-question-hook-scope.ts', 'test/helpers/skill-census.ts', 'test/plan-skill-question-hook-scope.test.ts', 'scripts/resolvers/testing.ts', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/review.ts', 'test/plan-review-cases.test.ts', 'scripts/resolvers/tasks-section.ts' @@ -301,11 +301,11 @@ export const E2E_TOUCHFILES: Record = { // transitively by the entries above). Two new standalone files exist for // skills with no prior plan-mode test: 'office-hours-auto-mode': [ - 'test/auto-decide-recommendation-scope.test.ts', + 'test/fixtures/auto-decide-recommendation-361c.json', - 'test/auto-decide-target-identity.test.ts', + 'test/fixtures/auto-decide-target-361c.json', -'test/native-auto-decide.test.ts', 'test/native-auto-decide-pty.test.ts', 'test/fixtures/native-auto-decide-ag.json', 'test/helpers/native-auto-decide.ts', 'test/auto-decide-current-declaration.test.ts', 'test/fixtures/auto-decide-current-declaration-6aef.json', 'test/auto-decide-explanatory-mode.test.ts', 'test/fixtures/auto-decide-explanatory-mode-043a.json', 'test/fixtures/auto-decide-explanatory-mode-749df.json', 'bin/gstack-skill-start', 'bin/gstack-skill-end', 'office-hours/**', 'scripts/resolvers/preamble/generate-completion-status.ts', 'scripts/resolvers/question-tuning.ts', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/preamble.ts', 'test/helpers/claude-pty-runner.ts', 'test/plan-count-design-ui-recovery.test.ts', 'test/fixtures/design-ui-boxed-question.json', 'test/helpers/hermetic-skill-runtime.ts', 'test/hermetic-skill-runtime.test.ts', 'test/helpers/pty-trust-dialog.ts', 'test/pty-trust-dialog.test.ts', 'test/skill-e2e-office-hours-auto-mode.test.ts', 'test/plan-count-truncated-border.test.ts', 'test/fixtures/eng-d2-truncated-border-0bcd.json', 'test/plan-count-truncated-question.test.ts', 'test/plan-count-clipped-elision.test.ts', 'test/fixtures/eng-d1-clipped-elision-1579.json', 'test/fixtures/eng-d2-planning-prelude-4d.json', 'test/fixtures/ceo-approach-z-call.json', 'test/fixtures/ceo-approach-z-screen.txt', +'test/native-auto-decide.test.ts', 'test/native-auto-decide-pty.test.ts', 'test/fixtures/native-auto-decide-ag.json', 'test/helpers/native-auto-decide.ts','test/fixtures/auto-decide-current-declaration-6aef.json','test/fixtures/auto-decide-explanatory-mode-043a.json', 'test/fixtures/auto-decide-explanatory-mode-749df.json', 'bin/gstack-skill-start', 'bin/gstack-skill-end', 'office-hours/**', 'scripts/resolvers/preamble/generate-completion-status.ts', 'scripts/resolvers/question-tuning.ts', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/preamble.ts', 'test/helpers/claude-pty-runner.ts', 'test/plan-count-design-ui-recovery.test.ts', 'test/fixtures/design-ui-boxed-question.json', 'test/helpers/hermetic-skill-runtime.ts', 'test/hermetic-skill-runtime.test.ts', 'test/helpers/pty-trust-dialog.ts', 'test/pty-trust-dialog.test.ts', 'test/skill-e2e-office-hours-auto-mode.test.ts', 'test/plan-count-truncated-border.test.ts', 'test/fixtures/eng-d2-truncated-border-0bcd.json', 'test/plan-count-truncated-question.test.ts', 'test/plan-count-clipped-elision.test.ts', 'test/fixtures/eng-d1-clipped-elision-1579.json', 'test/fixtures/eng-d2-planning-prelude-4d.json', 'test/fixtures/ceo-approach-z-call.json', 'test/fixtures/ceo-approach-z-screen.txt', 'test/pty-workspace-trust.test.ts', 'test/fixtures/pty-companion-cli.ts', 'lib/fs-atomic.ts', 'test/helpers/plan-skill-questions.ts', 'test/fixtures/eng-auq-validation-error.json', 'test/fixtures/bash-directory-permission.json', 'test/fixtures/design-tasks-bash-permission.json', 'test/plan-skill-read-permission.test.ts', 'test/fixtures/read-permission.json', 'test/pty-numbered-option-indent-native.test.ts', 'test/fixtures/ceo-split-e5-numbered-description-491.json', 'test/plan-skill-questions.test.ts', 'test/helpers/plan-skill-question-events.ts', 'test/plan-skill-question-events.test.ts', 'test/helpers/plan-skill-question-hook-scope.ts', 'test/helpers/skill-census.ts', 'test/plan-skill-question-hook-scope.test.ts' ], 'office-hours-phase4-fork': ['test/session-runner-stream-lifecycle.test.ts', 'bin/gstack-skill-start', 'bin/gstack-skill-end', 'office-hours/**', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/preamble/generate-completion-status.ts', 'scripts/resolvers/preamble.ts', 'scripts/resolvers/question-tuning.ts', 'test/helpers/llm-judge.ts', 'test/skill-e2e-office-hours-phase4.test.ts', 'test/office-hours-phase4-caller.test.ts'], @@ -317,15 +317,15 @@ export const E2E_TOUCHFILES: Record = { // infrastructure plus the resolvers that own the AUTO_DECIDE preamble. 'auto-decide-preserved': [ 'lib/claude-public-transcript.ts', - 'test/auto-decide-recommendation-scope.test.ts', + 'test/fixtures/auto-decide-recommendation-361c.json', - 'test/auto-decide-target-identity.test.ts', + 'test/fixtures/auto-decide-target-361c.json', 'test/paid-retry-supervision.test.ts', 'test/autoplan-public-narration.test.ts', 'test/fixtures/autoplan-public-narration-ad.json', 'test/fixtures/auto-decide-completed-mode-f359.json', 'test/helpers/plan-count-transcript.ts', 'test/plan-count-cross-cwd-ancestry.test.ts', 'test/fixtures/plan-count-cross-cwd-ancestry-0bcd.json', 'test/plan-count-session-cwd.test.ts','test/pty-screen-unicode-ap.test.ts', - 'test/auto-decide-saved-ai.test.ts', 'test/fixtures/auto-decide-saved-ai.json', 'test/fixtures/auto-decide-retry-ai.json','bin/gstack-skill-start', 'bin/gstack-skill-end', 'bin/gstack-session-kind', 'scripts/resolvers/question-tuning.ts', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/preamble/generate-preamble-bash.ts', 'scripts/resolvers/preamble/generate-completion-status.ts', 'plan-ceo-review/**', 'bin/gstack-question-preference', 'bin/gstack-config', 'bin/gstack-slug', 'hosts/claude/hooks/question-preference-hook.ts', 'hosts/claude/hooks/spawned-directive.ts', 'lib/is-conductor.ts', 'test/helpers/claude-pty-runner.ts', 'test/plan-count-design-ui-recovery.test.ts', 'test/fixtures/design-ui-boxed-question.json', 'test/helpers/hermetic-skill-runtime.ts', 'test/hermetic-skill-runtime.test.ts', 'test/helpers/pty-trust-dialog.ts', 'test/pty-trust-dialog.test.ts', 'test/skill-e2e-auto-decide-preserved.test.ts', 'test/plan-count-truncated-border.test.ts', 'test/fixtures/eng-d2-truncated-border-0bcd.json', 'test/plan-count-truncated-question.test.ts', 'test/plan-count-clipped-elision.test.ts', 'test/fixtures/eng-d1-clipped-elision-1579.json', 'test/fixtures/eng-d2-planning-prelude-4d.json', 'test/fixtures/ceo-approach-z-call.json', 'test/fixtures/ceo-approach-z-screen.txt', 'test/helpers/native-auto-decide.ts', 'test/auto-decide-current-declaration.test.ts', 'test/fixtures/auto-decide-current-declaration-6aef.json', 'test/auto-decide-explanatory-mode.test.ts', 'test/fixtures/auto-decide-explanatory-mode-043a.json', 'test/fixtures/auto-decide-explanatory-mode-749df.json', 'test/auto-decide-structured.test.ts', 'test/fixtures/auto-decide-structured-77.json', 'test/helpers/auto-decision-state.ts', 'test/auto-decision-state.test.ts', 'test/fixtures/auto-decide-state-cab3.json', 'bin/gstack-question-log', 'test/native-auto-decide.test.ts', 'test/native-auto-decide-pty.test.ts', 'test/helpers/fake-plan-seed.ts', 'test/helpers/plan-seed-submission.ts', 'test/plan-seed-submission.test.ts', 'test/fixtures/plan-seed-cli.ts', 'test/fixtures/native-auto-decide-ag.json', 'test/eng-seeded-completion-ai.test.ts', 'test/fixtures/eng-seeded-completion-ai.json', 'test/helpers/plan-count-pending-exit.ts', 'test/plan-count-pending-exit.test.ts', 'test/helpers/pty-screen.ts', 'test/pty-screen.test.ts', 'test/pty-screen-session.test.ts', 'test/fixtures/pty-screen/**', +'test/fixtures/auto-decide-saved-ai.json', 'test/fixtures/auto-decide-retry-ai.json','bin/gstack-skill-start', 'bin/gstack-skill-end', 'bin/gstack-session-kind', 'scripts/resolvers/question-tuning.ts', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/preamble/generate-preamble-bash.ts', 'scripts/resolvers/preamble/generate-completion-status.ts', 'plan-ceo-review/**', 'bin/gstack-question-preference', 'bin/gstack-config', 'bin/gstack-slug', 'hosts/claude/hooks/question-preference-hook.ts', 'hosts/claude/hooks/spawned-directive.ts', 'lib/is-conductor.ts', 'test/helpers/claude-pty-runner.ts', 'test/plan-count-design-ui-recovery.test.ts', 'test/fixtures/design-ui-boxed-question.json', 'test/helpers/hermetic-skill-runtime.ts', 'test/hermetic-skill-runtime.test.ts', 'test/helpers/pty-trust-dialog.ts', 'test/pty-trust-dialog.test.ts', 'test/skill-e2e-auto-decide-preserved.test.ts', 'test/plan-count-truncated-border.test.ts', 'test/fixtures/eng-d2-truncated-border-0bcd.json', 'test/plan-count-truncated-question.test.ts', 'test/plan-count-clipped-elision.test.ts', 'test/fixtures/eng-d1-clipped-elision-1579.json', 'test/fixtures/eng-d2-planning-prelude-4d.json', 'test/fixtures/ceo-approach-z-call.json', 'test/fixtures/ceo-approach-z-screen.txt', 'test/helpers/native-auto-decide.ts','test/fixtures/auto-decide-current-declaration-6aef.json','test/fixtures/auto-decide-explanatory-mode-043a.json', 'test/fixtures/auto-decide-explanatory-mode-749df.json','test/fixtures/auto-decide-structured-77.json', 'test/helpers/auto-decision-state.ts','test/fixtures/auto-decide-state-cab3.json', 'bin/gstack-question-log', 'test/native-auto-decide.test.ts', 'test/native-auto-decide-pty.test.ts', 'test/helpers/fake-plan-seed.ts', 'test/helpers/plan-seed-submission.ts', 'test/plan-seed-submission.test.ts', 'test/fixtures/plan-seed-cli.ts', 'test/fixtures/native-auto-decide-ag.json', 'test/eng-seeded-completion-ai.test.ts', 'test/fixtures/eng-seeded-completion-ai.json', 'test/helpers/plan-count-pending-exit.ts', 'test/plan-count-pending-exit.test.ts', 'test/helpers/pty-screen.ts', 'test/pty-screen.test.ts', 'test/pty-screen-session.test.ts', 'test/fixtures/pty-screen/**', "test/ceo-mode-preference-al.test.ts", 'test/helpers/plan-count-fixture.ts', 'test/plan-count-fixture.test.ts', 'test/pty-workspace-trust.test.ts', 'test/fixtures/pty-companion-cli.ts', 'lib/fs-atomic.ts', 'test/auto-decide-fixture.test.ts', 'test/fixtures/auto-decide-mode-selector-749df.json', 'lib/redact-engine.ts', 'lib/redact-patterns.ts', 'test/helpers/ceo-finding-fixture.ts', 'test/ceo-finding-fixture.test.ts', 'test/helpers/owned-claude-transcript.ts', 'test/helpers/plan-skill-questions.ts', 'test/fixtures/eng-auq-validation-error.json', 'test/fixtures/bash-directory-permission.json', 'test/fixtures/design-tasks-bash-permission.json', 'test/plan-skill-read-permission.test.ts', 'test/fixtures/read-permission.json', 'test/pty-numbered-option-indent-native.test.ts', 'test/fixtures/ceo-split-e5-numbered-description-491.json', 'test/plan-skill-questions.test.ts', 'test/helpers/plan-skill-question-events.ts', 'test/plan-skill-question-events.test.ts', 'test/helpers/plan-skill-question-hook-scope.ts', 'test/helpers/skill-census.ts', 'test/plan-skill-question-hook-scope.test.ts', 'scripts/resolvers/tasks-section.ts' @@ -357,8 +357,8 @@ export const E2E_TOUCHFILES: Record = { 'test/ceo-mode-pending-submit.test.ts', 'test/fixtures/ceo-mode-pending-submit.json', "test/plan-count-cross-cwd-ancestry.test.ts", "test/fixtures/plan-count-cross-cwd-ancestry-0bcd.json", "test/plan-count-session-cwd.test.ts", - "test/ceo-mode-colon-at.test.ts", "test/fixtures/ceo-mode-colon-at.json",'test/pty-screen-unicode-ap.test.ts', 'plan-ceo-review/**', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/preamble.ts', 'test/helpers/claude-pty-runner.ts', 'test/plan-count-design-ui-recovery.test.ts', 'test/fixtures/design-ui-boxed-question.json', 'test/plan-count-native-input.test.ts', 'test/helpers/pty-screen.ts', 'test/pty-screen.test.ts', 'test/pty-screen-session.test.ts', 'test/fixtures/pty-screen/**', 'test/helpers/hermetic-skill-runtime.ts', 'test/hermetic-skill-runtime.test.ts', 'test/helpers/pty-trust-dialog.ts', 'test/pty-trust-dialog.test.ts', 'test/helpers/ceo-mode-option.ts', 'test/ceo-mode-expansion-disposition.test.ts', 'test/fixtures/ceo-expansion-disposition-77.json', 'test/ceo-mode-option.test.ts', 'test/pty-option-selection.test.ts', 'test/helpers/plan-count-transcript.ts', 'test/autoplan-public-narration.test.ts', 'test/fixtures/autoplan-public-narration-ad.json', 'test/plan-count-transcript.test.ts', 'test/helpers/plan-count-fixture.ts', 'test/plan-count-fixture.test.ts', 'test/plan-count-prerequisite-n.test.ts', 'test/fixtures/ceo-prerequisite-n-call.json', 'test/fixtures/eng-prerequisite-77.json', 'test/skill-e2e-plan-ceo-mode-routing.test.ts', 'test/ceo-mode-posture-native.test.ts', 'test/fixtures/ceo-hold-posture-l.json', 'test/ceo-mode-labels-native.test.ts', 'test/fixtures/ceo-mode-labels-l.json', 'test/plan-count-checkbox.test.ts', 'test/fixtures/ceo-checkbox-l.screen.txt', 'test/ceo-mode-prerequisite.test.ts', 'test/fixtures/ceo-mode-prerequisite-o-calls.json', 'test/fixtures/ceo-mode-prerequisite-q-call.json', 'test/plan-count-preview-footer.test.ts', 'test/fixtures/ceo-preview-u-call.json', 'test/fixtures/ceo-preview-u-screen.txt', 'test/ceo-posture-packet.test.ts', 'test/fixtures/design-preview-v-screen.txt', 'test/plan-count-truncated-border.test.ts', 'test/fixtures/eng-d2-truncated-border-0bcd.json', 'test/plan-count-truncated-question.test.ts', 'test/plan-count-clipped-elision.test.ts', 'test/fixtures/eng-d1-clipped-elision-1579.json', 'test/fixtures/eng-d2-planning-prelude-4d.json', 'test/fixtures/ceo-approach-z-call.json', 'test/fixtures/ceo-approach-z-screen.txt', 'test/fixtures/ceo-mode-preview-aa-screen.txt', 'test/fixtures/ceo-count-mode-preview-aa-screen.txt', 'test/ceo-expansion-auq.test.ts', 'test/fixtures/ceo-expansion-auq-ac.json', 'test/helpers/plan-count-pending-question.ts', 'test/autoplan-pending-question.test.ts', 'test/plan-pending-question-pty.test.ts', 'test/ceo-barless-submit.test.ts', 'test/fixtures/ceo-barless-submit-ac.json', 'test/pending-question-completion.test.ts', 'test/fixtures/pending-question-completion-ad.json', 'test/ceo-mode-posture-ad.test.ts', 'test/fixtures/ceo-mode-posture-ad.json', 'test/ceo-mode-full-ad.test.ts', 'test/fixtures/ceo-mode-full-ad.json', 'test/ceo-prerequisite-ad-v2.test.ts', 'test/fixtures/ceo-prerequisite-ad-v2.json', 'test/ceo-hold-posture-ag.test.ts', 'test/fixtures/ceo-hold-posture-ag.json', - "test/ceo-hold-commitment-ar.test.ts", "test/fixtures/ceo-hold-commitment-ar.json", +"test/fixtures/ceo-mode-colon-at.json",'test/pty-screen-unicode-ap.test.ts', 'plan-ceo-review/**', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/preamble.ts', 'test/helpers/claude-pty-runner.ts', 'test/plan-count-design-ui-recovery.test.ts', 'test/fixtures/design-ui-boxed-question.json', 'test/plan-count-native-input.test.ts', 'test/helpers/pty-screen.ts', 'test/pty-screen.test.ts', 'test/pty-screen-session.test.ts', 'test/fixtures/pty-screen/**', 'test/helpers/hermetic-skill-runtime.ts', 'test/hermetic-skill-runtime.test.ts', 'test/helpers/pty-trust-dialog.ts', 'test/pty-trust-dialog.test.ts', 'test/helpers/ceo-mode-option.ts', 'test/ceo-mode-expansion-disposition.test.ts', 'test/fixtures/ceo-expansion-disposition-77.json', 'test/ceo-mode-option.test.ts', 'test/pty-option-selection.test.ts', 'test/helpers/plan-count-transcript.ts', 'test/autoplan-public-narration.test.ts', 'test/fixtures/autoplan-public-narration-ad.json', 'test/plan-count-transcript.test.ts', 'test/helpers/plan-count-fixture.ts', 'test/plan-count-fixture.test.ts', 'test/plan-count-prerequisite.test.ts', 'test/fixtures/ceo-prerequisite-n-call.json', 'test/fixtures/eng-prerequisite-77.json', 'test/skill-e2e-plan-ceo-mode-routing.test.ts', 'test/ceo-mode-posture-native.test.ts', 'test/fixtures/ceo-hold-posture-l.json', 'test/ceo-mode-labels-native.test.ts', 'test/fixtures/ceo-mode-labels-l.json', 'test/plan-count-checkbox.test.ts', 'test/fixtures/ceo-checkbox-l.screen.txt', 'test/ceo-mode-prerequisite.test.ts', 'test/fixtures/ceo-mode-prerequisite-o-calls.json', 'test/fixtures/ceo-mode-prerequisite-q-call.json', 'test/plan-count-preview-footer.test.ts', 'test/fixtures/ceo-preview-u-call.json', 'test/fixtures/ceo-preview-u-screen.txt', 'test/ceo-posture-packet.test.ts', 'test/fixtures/design-preview-v-screen.txt', 'test/plan-count-truncated-border.test.ts', 'test/fixtures/eng-d2-truncated-border-0bcd.json', 'test/plan-count-truncated-question.test.ts', 'test/plan-count-clipped-elision.test.ts', 'test/fixtures/eng-d1-clipped-elision-1579.json', 'test/fixtures/eng-d2-planning-prelude-4d.json', 'test/fixtures/ceo-approach-z-call.json', 'test/fixtures/ceo-approach-z-screen.txt', 'test/fixtures/ceo-mode-preview-aa-screen.txt', 'test/fixtures/ceo-count-mode-preview-aa-screen.txt', 'test/ceo-expansion-auq.test.ts', 'test/fixtures/ceo-expansion-auq-ac.json', 'test/helpers/plan-count-pending-question.ts', 'test/autoplan-pending-question.test.ts', 'test/plan-pending-question-pty.test.ts', 'test/ceo-barless-submit.test.ts', 'test/fixtures/ceo-barless-submit-ac.json', 'test/pending-question-completion.test.ts', 'test/fixtures/pending-question-completion-ad.json','test/fixtures/ceo-mode-posture-ad.json','test/fixtures/ceo-mode-full-ad.json','test/fixtures/ceo-prerequisite-ad-v2.json','test/fixtures/ceo-hold-posture-ag.json', +"test/fixtures/ceo-hold-commitment-ar.json", 'test/pty-workspace-trust.test.ts', 'test/fixtures/pty-companion-cli.ts', 'lib/fs-atomic.ts', 'test/ceo-mode-routing-fixture.test.ts', 'test/helpers/ceo-finding-fixture.ts', 'test/ceo-finding-fixture.test.ts', 'test/helpers/claude-pty-runner.unit.test.ts', 'test/helpers/plan-skill-questions.ts', 'test/fixtures/eng-auq-validation-error.json', 'test/fixtures/bash-directory-permission.json', 'test/fixtures/design-tasks-bash-permission.json', 'test/plan-skill-read-permission.test.ts', 'test/fixtures/read-permission.json', 'test/pty-numbered-option-indent-native.test.ts', 'test/fixtures/ceo-split-e5-numbered-description-491.json', 'test/plan-skill-questions.test.ts', 'test/helpers/plan-skill-question-events.ts', 'test/plan-skill-question-events.test.ts', 'test/helpers/plan-skill-question-hook-scope.ts', 'test/helpers/skill-census.ts', 'test/plan-skill-question-hook-scope.test.ts', 'test/helpers/owned-claude-transcript.ts', 'scripts/resolvers/tasks-section.ts', 'test/fixtures/ceo-expansion-pacing-77.json', ], @@ -366,12 +366,12 @@ export const E2E_TOUCHFILES: Record = { 'lib/claude-public-transcript.ts', 'test/autoplan-public-narration.test.ts', 'test/fixtures/autoplan-public-narration-ad.json', 'test/helpers/plan-count-fixture.ts', 'test/plan-count-fixture.test.ts', - 'test/plan-count-prerequisite-n.test.ts', 'test/fixtures/ceo-prerequisite-n-call.json', 'test/fixtures/eng-prerequisite-77.json', + 'test/plan-count-prerequisite.test.ts', 'test/fixtures/ceo-prerequisite-n-call.json', 'test/fixtures/eng-prerequisite-77.json', 'test/helpers/plan-count-transcript.ts', 'test/plan-count-cross-cwd-ancestry.test.ts', 'test/fixtures/plan-count-cross-cwd-ancestry-0bcd.json', 'test/plan-count-session-cwd.test.ts', - "test/plan-scope-recovery-av.test.ts", + "test/fixtures/plan-scope-recovery-av.json", "test/fixtures/design-scope-checkpoint-at.json",'plan-design-review/**', 'test/fixtures/plans/ui-heavy-feature.md', 'test/helpers/claude-pty-runner.ts', 'test/plan-count-design-ui-recovery.test.ts', 'test/fixtures/design-ui-boxed-question.json', 'test/helpers/hermetic-skill-runtime.ts', 'test/hermetic-skill-runtime.test.ts', 'test/helpers/pty-trust-dialog.ts', 'test/pty-trust-dialog.test.ts', 'test/skill-e2e-plan-design-with-ui.test.ts', 'test/plan-count-truncated-border.test.ts', 'test/fixtures/eng-d2-truncated-border-0bcd.json', 'test/plan-count-truncated-question.test.ts', 'test/plan-count-clipped-elision.test.ts', 'test/fixtures/eng-d1-clipped-elision-1579.json', 'test/fixtures/eng-d2-planning-prelude-4d.json', 'test/fixtures/ceo-approach-z-call.json', 'test/fixtures/ceo-approach-z-screen.txt', - "test/design-scope-entry-aq.test.ts", + "test/plan-scope-selection.test.ts", "test/review-entry-and-design-clarity-au.test.ts", "scripts/resolvers/preamble/generate-preamble-bash.ts", "scripts/resolvers/preamble/generate-completion-status.ts", 'test/pty-workspace-trust.test.ts', 'test/fixtures/pty-companion-cli.ts', 'lib/fs-atomic.ts', 'test/plan-design-with-ui-fixture.test.ts', 'test/helpers/ceo-finding-fixture.ts', 'test/helpers/plan-review-cases.ts', 'test/helpers/plan-review-board-feedback.ts', 'test/plan-review-board-feedback.test.ts', 'test/fixtures/design-board-questions.json', 'test/fixtures/design-outside-voices-question.json', 'design/src/daemon-state.ts', 'design/src/daemon.ts', 'design/test/daemon-tests-fixtures.ts', 'design/src/daemon-client.ts', 'test/helpers/owned-claude-transcript.ts', 'test/helpers/plan-skill-questions.ts', 'test/fixtures/eng-auq-validation-error.json', 'test/fixtures/bash-directory-permission.json', 'test/fixtures/design-tasks-bash-permission.json', 'test/plan-skill-read-permission.test.ts', 'test/fixtures/read-permission.json', 'test/pty-numbered-option-indent-native.test.ts', 'test/fixtures/ceo-split-e5-numbered-description-491.json', 'test/plan-skill-questions.test.ts', 'test/helpers/plan-skill-question-events.ts', 'test/plan-skill-question-events.test.ts', 'test/helpers/plan-skill-question-hook-scope.ts', 'test/helpers/skill-census.ts', 'test/plan-skill-question-hook-scope.test.ts', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'bin/gstack-paths', 'bin/gstack-slug', 'scripts/resolvers/design.ts' @@ -395,8 +395,8 @@ export const E2E_TOUCHFILES: Record = { 'test/section-capture-native-tools.test.ts', 'test/ship-section-fixture.test.ts', 'scripts/resolvers/testing.ts' ], 'plan-ceo-section-loading': ['test/session-runner-stream-lifecycle.test.ts', - 'test/fixtures/ceo-fill-lifetime.json','test/gstack-brain-context-load.test.ts', 'plan-ceo-review/**', 'test/skill-ceo-section-ordering.test.ts', 'scripts/resolvers/sections.ts', 'scripts/gen-skill-docs.ts', 'test/helpers/auq-sdk-capture.ts', 'test/helpers/session-runner.ts', 'test/session-runner-tools.test.ts', 'test/helpers/ceo-section-loading-fixture.ts', 'test/ceo-section-loading-fixture.test.ts', 'test/skill-e2e-plan-ceo-review-section-loading.test.ts', 'test/fixtures/ceo-section-loading-l-report.md', 'test/fixtures/ceo-section-loading-q-report.md', 'test/fixtures/ceo-section-r-rejected-report.md', 'test/fixtures/ceo-section-s-trace-report.md', 'test/fixtures/ceo-section-u-report.md', 'test/fixtures/ceo-section-y-report.md', 'test/fixtures/ceo-section-aa-report.md', 'test/sdk-stale-table-ad-v3.test.ts', 'test/fixtures/sdk-stale-table-ad-v3.json', 'test/sdk-ordering-ae.test.ts', 'test/fixtures/sdk-ordering-ae.json', 'test/sdk-columnar-af.test.ts', 'test/fixtures/sdk-columnar-af.json', 'test/sdk-order-b-ag.test.ts', 'test/fixtures/sdk-order-b-ag.json', 'test/sdk-schedule-continuation-ah.test.ts', 'test/fixtures/sdk-schedule-continuation-ah.json', 'test/sdk-original-order-ai.test.ts', 'test/fixtures/sdk-original-order-ai.json', 'test/sdk-compact-sequence-aj.test.ts', 'test/fixtures/sdk-compact-sequence-aj.json', - "test/sdk-reported-coordination-ar.test.ts", "test/fixtures/sdk-reported-coordination-ar.md", "test/sdk-ordered-schedule-ar.test.ts", "test/fixtures/sdk-ordered-schedule-ar.md", + 'test/fixtures/ceo-fill-lifetime.json','test/gstack-brain-context-load.test.ts', 'plan-ceo-review/**', 'test/skill-ceo-section-ordering.test.ts', 'scripts/resolvers/sections.ts', 'scripts/gen-skill-docs.ts', 'test/helpers/auq-sdk-capture.ts', 'test/helpers/session-runner.ts', 'test/session-runner-tools.test.ts', 'test/helpers/ceo-section-loading-fixture.ts', 'test/ceo-section-loading-fixture.test.ts', 'test/skill-e2e-plan-ceo-review-section-loading.test.ts', 'test/fixtures/ceo-section-loading-l-report.md', 'test/fixtures/ceo-section-loading-q-report.md', 'test/fixtures/ceo-section-r-rejected-report.md', 'test/fixtures/ceo-section-s-trace-report.md', 'test/fixtures/ceo-section-u-report.md', 'test/fixtures/ceo-section-y-report.md', 'test/fixtures/ceo-section-aa-report.md','test/fixtures/sdk-stale-table-ad-v3.json','test/fixtures/sdk-ordering-ae.json','test/fixtures/sdk-columnar-af.json','test/fixtures/sdk-order-b-ag.json','test/fixtures/sdk-schedule-continuation-ah.json','test/fixtures/sdk-original-order-ai.json','test/fixtures/sdk-compact-sequence-aj.json', +"test/fixtures/sdk-reported-coordination-ar.md","test/fixtures/sdk-ordered-schedule-ar.md", 'test/section-capture-native-tools.test.ts', 'scripts/resolvers/review.ts', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/tasks-section.ts' ], // Data-driven behavioral guard for the 'plan'/'prompt' carves (eng, design, @@ -408,10 +408,10 @@ export const E2E_TOUCHFILES: Record = { 'bin/gstack-slug', 'bin/gstack-wtree', 'bin/gstack-config', 'bin/gstack-brain-enqueue', 'test/autoplan-amend-input.test.ts', 'test/fixtures/autoplan-amend-input-77.json', 'test/autoplan-phase-handoff.test.ts', 'test/fixtures/autoplan-phase-handoff-6714.json','scripts/resolvers/learnings.ts', 'test/gstack-paths.test.ts', - "test/plan-scope-recovery-av.test.ts", + "test/fixtures/plan-scope-recovery-av.json", "test/fixtures/design-scope-checkpoint-at.json",'test/eng-scope-entry-ap.test.ts', 'scripts/resolvers/composition.ts', 'test/autoplan-review-discovery.test.ts', 'test/autoplan-phase-order.test.ts', 'design-html/**', 'design-shotgun/**', 'qa/**', 'browse/**', 'retro/**', 'autoplan/**', 'spec/**', 'setup-gbrain/**', 'review/**', 'codex/**', 'land-and-deploy/**', 'plan-eng-review/**', 'plan-design-review/**', 'plan-devex-review/**', 'document-release/**', 'design-consultation/**', 'test/helpers/carve-guards.ts', 'scripts/resolvers/sections.ts', 'scripts/resolvers/redact-doc.ts', 'scripts/gen-skill-docs.ts', 'test/helpers/auq-sdk-capture.ts', 'test/helpers/session-runner.ts', 'test/session-runner-tools.test.ts', 'bin/gstack-autoplan-snapshot.ts', 'test/autoplan-snapshot.test.ts', 'test/autoplan-init.test.ts', 'test/autoplan-obligations.test.ts', 'test/fixtures/autoplan/t-ceo-omitted-obligations.json', 'test/fixtures/autoplan/u-ceo-original-loss.json', 'test/fixtures/autoplan/v-ceo-dangling-references.json', - "test/design-scope-entry-aq.test.ts", + "test/plan-scope-selection.test.ts", "test/review-entry-and-design-clarity-au.test.ts", "scripts/resolvers/preamble/generate-preamble-bash.ts", "scripts/resolvers/preamble/generate-completion-status.ts", 'test/section-capture-native-tools.test.ts', 'test/carve-section-loading*.test.ts', 'test/helpers/carve-section-case.ts', 'test/codex-carve-fixture.test.ts', 'test/carve-section-sharding.test.ts', 'test/carve-section-loading-browse.test.ts', 'test/carve-section-loading-codex.test.ts', 'test/carve-section-loading-design-consultation.test.ts', 'test/carve-section-loading-design-html.test.ts', 'test/carve-section-loading-design-shotgun.test.ts', 'test/carve-section-loading-document-release.test.ts', 'test/carve-section-loading-land-and-deploy.test.ts', 'test/carve-section-loading-plan-design-review.test.ts', 'test/carve-section-loading-plan-devex-review.test.ts', 'test/carve-section-loading-plan-eng-review.test.ts', 'test/carve-section-loading-qa.test.ts', 'test/carve-section-loading-retro.test.ts', 'test/carve-section-loading-review.test.ts', 'test/carve-section-loading-setup-gbrain.test.ts', 'test/carve-section-loading-spec.test.ts', 'test/design-html-section-completion.test.ts', 'test/fixtures/design-html-section-complete.md', 'scripts/resolvers/testing.ts', 'test/helpers/carve-plan-fixture.ts', 'test/carve-plan-fixture.test.ts', 'test/fixtures/carve-existing-repository/**', 'scripts/resolvers/review.ts', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'test/plan-review-cases.test.ts' @@ -435,11 +435,11 @@ export const E2E_TOUCHFILES: Record = { 'test/plan-floor-review.test.ts', 'test/fixtures/plan-floor-routing-361c.json', - 'test/fixtures/ceo-report-permission-fb10.json', 'test/plan-floor-permission.test.ts', 'test/fixtures/plan-floor-permission-fb10.json', 'test/helpers/plan-count-file-permission.ts', 'test/plan-edit-cropped-permission.test.ts', 'test/fixtures/plan-edit-cropped-permission-1579.json', 'test/plan-count-file-permission.test.ts', 'test/fixtures/plan-count-permission-target-ad-v2.json', 'test/design-crop-gutter-ap.test.ts', 'test/fixtures/design-crop-gutter-ap.json', 'test/plan-count-crop-ak.test.ts', 'test/fixtures/plan-count-crop-ak.json', 'test/plan-count-long-edit.test.ts', 'test/fixtures/plan-count-long-edit-0bcd.json', 'test/plan-count-cropped-wrap.test.ts', 'test/fixtures/plan-count-cropped-wrap-6714.json','test/paid-retry-supervision.test.ts', + 'test/fixtures/ceo-report-permission-fb10.json', 'test/plan-floor-permission.test.ts', 'test/fixtures/plan-floor-permission-fb10.json', 'test/helpers/plan-count-file-permission.ts', 'test/plan-edit-cropped-permission.test.ts', 'test/fixtures/plan-edit-cropped-permission-1579.json', 'test/plan-count-file-permission.test.ts', 'test/fixtures/plan-count-permission-target-ad-v2.json','test/fixtures/design-crop-gutter-ap.json','test/fixtures/plan-count-crop-ak.json', 'test/plan-count-long-edit.test.ts', 'test/fixtures/plan-count-long-edit-0bcd.json', 'test/plan-count-cropped-wrap.test.ts', 'test/fixtures/plan-count-cropped-wrap-6714.json','test/paid-retry-supervision.test.ts', 'scripts/resolvers/learnings.ts', - "test/plan-scope-recovery-av.test.ts", - "test/fixtures/plan-scope-recovery-av.json", 'test/eng-scope-entry-ap.test.ts', 'bin/gstack-skill-start', 'bin/gstack-skill-end', 'plan-eng-review/**', 'scripts/resolvers/preamble.ts', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/preamble/generate-completion-status.ts', 'scripts/resolvers/review.ts', 'test/helpers/claude-pty-runner.ts', 'test/plan-count-design-ui-recovery.test.ts', 'test/fixtures/design-ui-boxed-question.json', 'test/helpers/hermetic-skill-runtime.ts', 'test/hermetic-skill-runtime.test.ts', 'test/helpers/pty-trust-dialog.ts', 'test/pty-trust-dialog.test.ts', 'test/fixtures/forcing-finding-seeds.ts', 'test/skill-e2e-plan-eng-finding-floor.test.ts', 'test/eng-first-review-t.test.ts', 'test/fixtures/eng-batching-t-calls.json', 'test/eng-scope-y.test.ts', 'test/fixtures/eng-scope-y-calls.json', 'test/plan-count-truncated-border.test.ts', 'test/fixtures/eng-d2-truncated-border-0bcd.json', 'test/plan-count-truncated-question.test.ts', 'test/plan-count-clipped-elision.test.ts', 'test/fixtures/eng-d1-clipped-elision-1579.json', 'test/fixtures/eng-d2-planning-prelude-4d.json', 'test/fixtures/ceo-approach-z-call.json', 'test/fixtures/ceo-approach-z-screen.txt', 'test/eng-binding-z.test.ts', 'test/fixtures/eng-binding-z-calls.json', 'test/eng-binding-retry-z.test.ts', 'test/fixtures/eng-binding-retry-z-calls.json', 'test/helpers/plan-floor-target.ts', 'test/plan-floor-target.test.ts', 'test/helpers/plan-count-fixture.ts', 'test/plan-count-fixture.test.ts', 'test/helpers/plan-count-artifacts.ts', 'test/plan-count-artifacts.test.ts', - "test/eng-injected-export-aq.test.ts", "test/fixtures/eng-injected-export-aq.json", "test/eng-library-hooks-aq.test.ts", "test/fixtures/eng-library-hooks-aq.json", + "test/plan-scope-selection.test.ts", + "test/fixtures/plan-scope-recovery-av.json", 'test/eng-scope-entry-ap.test.ts', 'bin/gstack-skill-start', 'bin/gstack-skill-end', 'plan-eng-review/**', 'scripts/resolvers/preamble.ts', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/preamble/generate-completion-status.ts', 'scripts/resolvers/review.ts', 'test/helpers/claude-pty-runner.ts', 'test/plan-count-design-ui-recovery.test.ts', 'test/fixtures/design-ui-boxed-question.json', 'test/helpers/hermetic-skill-runtime.ts', 'test/hermetic-skill-runtime.test.ts', 'test/helpers/pty-trust-dialog.ts', 'test/pty-trust-dialog.test.ts', 'test/fixtures/forcing-finding-seeds.ts', 'test/skill-e2e-plan-eng-finding-floor.test.ts','test/fixtures/eng-batching-t-calls.json','test/fixtures/eng-scope-y-calls.json', 'test/plan-count-truncated-border.test.ts', 'test/fixtures/eng-d2-truncated-border-0bcd.json', 'test/plan-count-truncated-question.test.ts', 'test/plan-count-clipped-elision.test.ts', 'test/fixtures/eng-d1-clipped-elision-1579.json', 'test/fixtures/eng-d2-planning-prelude-4d.json', 'test/fixtures/ceo-approach-z-call.json', 'test/fixtures/ceo-approach-z-screen.txt','test/fixtures/eng-binding-z-calls.json', 'test/eng-first-review.test.ts', 'test/fixtures/eng-binding-retry-z-calls.json', 'test/helpers/plan-floor-target.ts', 'test/plan-floor-target.test.ts', 'test/helpers/plan-count-fixture.ts', 'test/plan-count-fixture.test.ts', 'test/helpers/plan-count-artifacts.ts', 'test/plan-count-artifacts.test.ts', +"test/fixtures/eng-injected-export-aq.json","test/fixtures/eng-library-hooks-aq.json", "test/review-entry-and-design-clarity-au.test.ts", "scripts/resolvers/preamble/generate-preamble-bash.ts", 'test/pty-workspace-trust.test.ts', 'test/fixtures/pty-companion-cli.ts', 'lib/fs-atomic.ts', 'test/helpers/plan-skill-questions.ts', 'test/fixtures/eng-auq-validation-error.json', 'test/fixtures/bash-directory-permission.json', 'test/fixtures/design-tasks-bash-permission.json', 'test/plan-skill-read-permission.test.ts', 'test/fixtures/read-permission.json', 'test/pty-numbered-option-indent-native.test.ts', 'test/fixtures/ceo-split-e5-numbered-description-491.json', 'test/plan-skill-questions.test.ts', 'test/helpers/plan-skill-question-events.ts', 'test/plan-skill-question-events.test.ts', 'test/helpers/plan-skill-question-hook-scope.ts', 'test/helpers/skill-census.ts', 'test/plan-skill-question-hook-scope.test.ts', 'scripts/resolvers/testing.ts', 'test/plan-review-cases.test.ts' @@ -452,7 +452,7 @@ export const E2E_TOUCHFILES: Record = { 'test/plan-floor-review.test.ts', 'test/fixtures/plan-floor-routing-361c.json', - 'test/fixtures/ceo-report-permission-fb10.json', 'test/plan-floor-permission.test.ts', 'test/fixtures/plan-floor-permission-fb10.json', 'test/helpers/plan-count-file-permission.ts', 'test/plan-edit-cropped-permission.test.ts', 'test/fixtures/plan-edit-cropped-permission-1579.json', 'test/plan-count-file-permission.test.ts', 'test/fixtures/plan-count-permission-target-ad-v2.json', 'test/design-crop-gutter-ap.test.ts', 'test/fixtures/design-crop-gutter-ap.json', 'test/plan-count-crop-ak.test.ts', 'test/fixtures/plan-count-crop-ak.json', 'test/plan-count-long-edit.test.ts', 'test/fixtures/plan-count-long-edit-0bcd.json', 'test/plan-count-cropped-wrap.test.ts', 'test/fixtures/plan-count-cropped-wrap-6714.json','test/paid-retry-supervision.test.ts', 'bin/gstack-skill-start', 'bin/gstack-skill-end', 'plan-ceo-review/**', 'scripts/resolvers/preamble.ts', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/preamble/generate-completion-status.ts', 'scripts/resolvers/review.ts', 'test/helpers/claude-pty-runner.ts', 'test/plan-count-design-ui-recovery.test.ts', 'test/fixtures/design-ui-boxed-question.json', 'test/helpers/hermetic-skill-runtime.ts', 'test/hermetic-skill-runtime.test.ts', 'test/helpers/pty-trust-dialog.ts', 'test/pty-trust-dialog.test.ts', 'test/fixtures/forcing-finding-seeds.ts', 'test/skill-e2e-plan-ceo-finding-floor.test.ts', 'test/plan-count-truncated-border.test.ts', 'test/fixtures/eng-d2-truncated-border-0bcd.json', 'test/plan-count-truncated-question.test.ts', 'test/plan-count-clipped-elision.test.ts', 'test/fixtures/eng-d1-clipped-elision-1579.json', 'test/fixtures/eng-d2-planning-prelude-4d.json', 'test/fixtures/ceo-approach-z-call.json', 'test/fixtures/ceo-approach-z-screen.txt', 'test/helpers/plan-floor-target.ts', 'test/plan-floor-target.test.ts', 'test/helpers/plan-count-fixture.ts', 'test/plan-count-fixture.test.ts', 'test/helpers/plan-count-artifacts.ts', 'test/plan-count-artifacts.test.ts', + 'test/fixtures/ceo-report-permission-fb10.json', 'test/plan-floor-permission.test.ts', 'test/fixtures/plan-floor-permission-fb10.json', 'test/helpers/plan-count-file-permission.ts', 'test/plan-edit-cropped-permission.test.ts', 'test/fixtures/plan-edit-cropped-permission-1579.json', 'test/plan-count-file-permission.test.ts', 'test/fixtures/plan-count-permission-target-ad-v2.json','test/fixtures/design-crop-gutter-ap.json','test/fixtures/plan-count-crop-ak.json', 'test/plan-count-long-edit.test.ts', 'test/fixtures/plan-count-long-edit-0bcd.json', 'test/plan-count-cropped-wrap.test.ts', 'test/fixtures/plan-count-cropped-wrap-6714.json','test/paid-retry-supervision.test.ts', 'bin/gstack-skill-start', 'bin/gstack-skill-end', 'plan-ceo-review/**', 'scripts/resolvers/preamble.ts', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/preamble/generate-completion-status.ts', 'scripts/resolvers/review.ts', 'test/helpers/claude-pty-runner.ts', 'test/plan-count-design-ui-recovery.test.ts', 'test/fixtures/design-ui-boxed-question.json', 'test/helpers/hermetic-skill-runtime.ts', 'test/hermetic-skill-runtime.test.ts', 'test/helpers/pty-trust-dialog.ts', 'test/pty-trust-dialog.test.ts', 'test/fixtures/forcing-finding-seeds.ts', 'test/skill-e2e-plan-ceo-finding-floor.test.ts', 'test/plan-count-truncated-border.test.ts', 'test/fixtures/eng-d2-truncated-border-0bcd.json', 'test/plan-count-truncated-question.test.ts', 'test/plan-count-clipped-elision.test.ts', 'test/fixtures/eng-d1-clipped-elision-1579.json', 'test/fixtures/eng-d2-planning-prelude-4d.json', 'test/fixtures/ceo-approach-z-call.json', 'test/fixtures/ceo-approach-z-screen.txt', 'test/helpers/plan-floor-target.ts', 'test/plan-floor-target.test.ts', 'test/helpers/plan-count-fixture.ts', 'test/plan-count-fixture.test.ts', 'test/helpers/plan-count-artifacts.ts', 'test/plan-count-artifacts.test.ts', 'test/pty-workspace-trust.test.ts', 'test/fixtures/pty-companion-cli.ts', 'lib/fs-atomic.ts', 'test/helpers/plan-skill-questions.ts', 'test/fixtures/eng-auq-validation-error.json', 'test/fixtures/bash-directory-permission.json', 'test/fixtures/design-tasks-bash-permission.json', 'test/plan-skill-read-permission.test.ts', 'test/fixtures/read-permission.json', 'test/pty-numbered-option-indent-native.test.ts', 'test/fixtures/ceo-split-e5-numbered-description-491.json', 'test/plan-skill-questions.test.ts', 'test/helpers/plan-skill-question-events.ts', 'test/plan-skill-question-events.test.ts', 'test/helpers/plan-skill-question-hook-scope.ts', 'test/helpers/skill-census.ts', 'test/plan-skill-question-hook-scope.test.ts', 'scripts/resolvers/tasks-section.ts' @@ -466,11 +466,11 @@ export const E2E_TOUCHFILES: Record = { 'test/plan-floor-review.test.ts', 'test/fixtures/plan-floor-routing-361c.json', - 'test/fixtures/ceo-report-permission-fb10.json', 'test/plan-floor-permission.test.ts', 'test/fixtures/plan-floor-permission-fb10.json', 'test/helpers/plan-count-file-permission.ts', 'test/plan-edit-cropped-permission.test.ts', 'test/fixtures/plan-edit-cropped-permission-1579.json', 'test/plan-count-file-permission.test.ts', 'test/fixtures/plan-count-permission-target-ad-v2.json', 'test/design-crop-gutter-ap.test.ts', 'test/fixtures/design-crop-gutter-ap.json', 'test/plan-count-crop-ak.test.ts', 'test/fixtures/plan-count-crop-ak.json', 'test/plan-count-long-edit.test.ts', 'test/fixtures/plan-count-long-edit-0bcd.json', 'test/plan-count-cropped-wrap.test.ts', 'test/fixtures/plan-count-cropped-wrap-6714.json', - "test/plan-scope-recovery-av.test.ts", + 'test/fixtures/ceo-report-permission-fb10.json', 'test/plan-floor-permission.test.ts', 'test/fixtures/plan-floor-permission-fb10.json', 'test/helpers/plan-count-file-permission.ts', 'test/plan-edit-cropped-permission.test.ts', 'test/fixtures/plan-edit-cropped-permission-1579.json', 'test/plan-count-file-permission.test.ts', 'test/fixtures/plan-count-permission-target-ad-v2.json','test/fixtures/design-crop-gutter-ap.json','test/fixtures/plan-count-crop-ak.json', 'test/plan-count-long-edit.test.ts', 'test/fixtures/plan-count-long-edit-0bcd.json', 'test/plan-count-cropped-wrap.test.ts', 'test/fixtures/plan-count-cropped-wrap-6714.json', + "test/fixtures/plan-scope-recovery-av.json", "test/fixtures/design-scope-checkpoint-at.json",'bin/gstack-skill-start', 'bin/gstack-skill-end', 'plan-design-review/**', 'scripts/resolvers/preamble.ts', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/preamble/generate-completion-status.ts', 'scripts/resolvers/review.ts', 'test/helpers/claude-pty-runner.ts', 'test/plan-count-design-ui-recovery.test.ts', 'test/fixtures/design-ui-boxed-question.json', 'test/helpers/hermetic-skill-runtime.ts', 'test/hermetic-skill-runtime.test.ts', 'test/helpers/pty-trust-dialog.ts', 'test/pty-trust-dialog.test.ts', 'test/fixtures/forcing-finding-seeds.ts', 'test/skill-e2e-plan-design-finding-floor.test.ts', 'test/plan-count-truncated-border.test.ts', 'test/fixtures/eng-d2-truncated-border-0bcd.json', 'test/plan-count-truncated-question.test.ts', 'test/plan-count-clipped-elision.test.ts', 'test/fixtures/eng-d1-clipped-elision-1579.json', 'test/fixtures/eng-d2-planning-prelude-4d.json', 'test/fixtures/ceo-approach-z-call.json', 'test/fixtures/ceo-approach-z-screen.txt', 'test/helpers/plan-floor-target.ts', 'test/plan-floor-target.test.ts', 'test/helpers/plan-count-fixture.ts', 'test/plan-count-fixture.test.ts', 'test/helpers/plan-count-artifacts.ts', 'test/plan-count-artifacts.test.ts', - "test/design-scope-entry-aq.test.ts", + "test/plan-scope-selection.test.ts", "test/review-entry-and-design-clarity-au.test.ts", "scripts/resolvers/preamble/generate-preamble-bash.ts", 'test/pty-workspace-trust.test.ts', 'test/fixtures/pty-companion-cli.ts', 'lib/fs-atomic.ts', 'test/helpers/plan-skill-questions.ts', 'test/fixtures/eng-auq-validation-error.json', 'test/fixtures/bash-directory-permission.json', 'test/fixtures/design-tasks-bash-permission.json', 'test/plan-skill-read-permission.test.ts', 'test/fixtures/read-permission.json', 'test/pty-numbered-option-indent-native.test.ts', 'test/fixtures/ceo-split-e5-numbered-description-491.json', 'test/plan-skill-questions.test.ts', 'test/helpers/plan-skill-question-events.ts', 'test/plan-skill-question-events.test.ts', 'test/helpers/plan-skill-question-hook-scope.ts', 'test/helpers/skill-census.ts', 'test/plan-skill-question-hook-scope.test.ts', 'test/helpers/ceo-finding-fixture.ts', 'test/ceo-finding-fixture.test.ts', 'test/plan-design-floor-fixture.test.ts' @@ -485,7 +485,7 @@ export const E2E_TOUCHFILES: Record = { 'test/plan-floor-review.test.ts', 'test/fixtures/plan-floor-routing-361c.json', - 'test/fixtures/ceo-report-permission-fb10.json', 'test/plan-floor-permission.test.ts', 'test/fixtures/plan-floor-permission-fb10.json', 'test/helpers/plan-count-file-permission.ts', 'test/plan-edit-cropped-permission.test.ts', 'test/fixtures/plan-edit-cropped-permission-1579.json', 'test/plan-count-file-permission.test.ts', 'test/fixtures/plan-count-permission-target-ad-v2.json', 'test/design-crop-gutter-ap.test.ts', 'test/fixtures/design-crop-gutter-ap.json', 'test/plan-count-crop-ak.test.ts', 'test/fixtures/plan-count-crop-ak.json', 'test/plan-count-long-edit.test.ts', 'test/fixtures/plan-count-long-edit-0bcd.json', 'test/plan-count-cropped-wrap.test.ts', 'test/fixtures/plan-count-cropped-wrap-6714.json','bin/gstack-skill-start', 'bin/gstack-skill-end', 'plan-devex-review/**', 'scripts/resolvers/preamble.ts', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/preamble/generate-completion-status.ts', 'scripts/resolvers/review.ts', 'test/helpers/claude-pty-runner.ts', 'test/plan-count-design-ui-recovery.test.ts', 'test/fixtures/design-ui-boxed-question.json', 'test/helpers/hermetic-skill-runtime.ts', 'test/hermetic-skill-runtime.test.ts', 'test/helpers/pty-trust-dialog.ts', 'test/pty-trust-dialog.test.ts', 'test/fixtures/forcing-finding-seeds.ts', 'test/skill-e2e-plan-devex-finding-floor.test.ts', 'test/plan-count-truncated-border.test.ts', 'test/fixtures/eng-d2-truncated-border-0bcd.json', 'test/plan-count-truncated-question.test.ts', 'test/plan-count-clipped-elision.test.ts', 'test/fixtures/eng-d1-clipped-elision-1579.json', 'test/fixtures/eng-d2-planning-prelude-4d.json', 'test/fixtures/ceo-approach-z-call.json', 'test/fixtures/ceo-approach-z-screen.txt', 'test/helpers/plan-floor-target.ts', 'test/plan-floor-target.test.ts', 'test/helpers/plan-count-fixture.ts', 'test/plan-count-fixture.test.ts', 'test/helpers/plan-count-artifacts.ts', 'test/plan-count-artifacts.test.ts', + 'test/fixtures/ceo-report-permission-fb10.json', 'test/plan-floor-permission.test.ts', 'test/fixtures/plan-floor-permission-fb10.json', 'test/helpers/plan-count-file-permission.ts', 'test/plan-edit-cropped-permission.test.ts', 'test/fixtures/plan-edit-cropped-permission-1579.json', 'test/plan-count-file-permission.test.ts', 'test/fixtures/plan-count-permission-target-ad-v2.json','test/fixtures/design-crop-gutter-ap.json','test/fixtures/plan-count-crop-ak.json', 'test/plan-count-long-edit.test.ts', 'test/fixtures/plan-count-long-edit-0bcd.json', 'test/plan-count-cropped-wrap.test.ts', 'test/fixtures/plan-count-cropped-wrap-6714.json','bin/gstack-skill-start', 'bin/gstack-skill-end', 'plan-devex-review/**', 'scripts/resolvers/preamble.ts', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/preamble/generate-completion-status.ts', 'scripts/resolvers/review.ts', 'test/helpers/claude-pty-runner.ts', 'test/plan-count-design-ui-recovery.test.ts', 'test/fixtures/design-ui-boxed-question.json', 'test/helpers/hermetic-skill-runtime.ts', 'test/hermetic-skill-runtime.test.ts', 'test/helpers/pty-trust-dialog.ts', 'test/pty-trust-dialog.test.ts', 'test/fixtures/forcing-finding-seeds.ts', 'test/skill-e2e-plan-devex-finding-floor.test.ts', 'test/plan-count-truncated-border.test.ts', 'test/fixtures/eng-d2-truncated-border-0bcd.json', 'test/plan-count-truncated-question.test.ts', 'test/plan-count-clipped-elision.test.ts', 'test/fixtures/eng-d1-clipped-elision-1579.json', 'test/fixtures/eng-d2-planning-prelude-4d.json', 'test/fixtures/ceo-approach-z-call.json', 'test/fixtures/ceo-approach-z-screen.txt', 'test/helpers/plan-floor-target.ts', 'test/plan-floor-target.test.ts', 'test/helpers/plan-count-fixture.ts', 'test/plan-count-fixture.test.ts', 'test/helpers/plan-count-artifacts.ts', 'test/plan-count-artifacts.test.ts', 'test/pty-workspace-trust.test.ts', 'test/fixtures/pty-companion-cli.ts', 'lib/fs-atomic.ts', 'test/helpers/plan-skill-questions.ts', 'test/fixtures/eng-auq-validation-error.json', 'test/fixtures/bash-directory-permission.json', 'test/fixtures/design-tasks-bash-permission.json', 'test/plan-skill-read-permission.test.ts', 'test/fixtures/read-permission.json', 'test/pty-numbered-option-indent-native.test.ts', 'test/fixtures/ceo-split-e5-numbered-description-491.json', 'test/plan-skill-questions.test.ts', 'test/helpers/plan-skill-question-events.ts', 'test/plan-skill-question-events.test.ts', 'test/helpers/plan-skill-question-hook-scope.ts', 'test/helpers/skill-census.ts', 'test/plan-skill-question-hook-scope.test.ts' ], @@ -517,16 +517,16 @@ export const E2E_TOUCHFILES: Record = { 'test/eng-resolution-block-position.test.ts', "test/plan-count-cross-cwd-ancestry.test.ts", "test/fixtures/plan-count-cross-cwd-ancestry-0bcd.json", "test/plan-count-session-cwd.test.ts", - "test/plan-scope-recovery-av.test.ts", + "test/plan-scope-selection.test.ts", "test/fixtures/plan-scope-recovery-av.json", - "test/batching-permission-at.test.ts", "test/fixtures/batching-permission-at.json", - "test/eng-declared-retry-at.test.ts", "test/fixtures/eng-declared-retry-at.json",'test/pty-screen-unicode-ap.test.ts', 'test/design-crop-gutter-ap.test.ts', 'test/fixtures/design-crop-gutter-ap.json', 'test/eng-scope-entry-ap.test.ts', 'bin/gstack-config', 'bin/gstack-skill-start', 'bin/gstack-skill-end', 'plan-eng-review/**', 'scripts/resolvers/preamble.ts', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/preamble/generate-completion-status.ts', 'scripts/resolvers/review.ts', 'test/helpers/claude-pty-runner.ts', 'test/plan-count-design-ui-recovery.test.ts', 'test/fixtures/design-ui-boxed-question.json', 'test/plan-count-native-input.test.ts', 'test/helpers/pty-screen.ts', 'test/pty-screen.test.ts', 'test/pty-screen-session.test.ts', 'test/fixtures/pty-screen/**', 'test/helpers/hermetic-skill-runtime.ts', 'test/hermetic-skill-runtime.test.ts', 'test/helpers/claude-pty-runner.unit.test.ts', 'test/helpers/pty-trust-dialog.ts', 'test/pty-trust-dialog.test.ts', 'test/helpers/plan-count-transcript.ts', 'test/autoplan-public-narration.test.ts', 'test/fixtures/autoplan-public-narration-ad.json', 'test/helpers/plan-count-pending-exit.ts', 'test/plan-count-pending-exit.test.ts', 'test/plan-count-transcript.test.ts', 'test/helpers/plan-count-artifacts.ts', 'test/plan-count-artifacts.test.ts', 'test/helpers/eval-store.ts', 'test/helpers/plan-count-fixture.ts', 'test/plan-count-fixture.test.ts', 'test/plan-count-collection-completion.test.ts', 'test/plan-count-timeout.test.ts', 'test/plan-count-navigation-r.test.ts', 'test/plan-count-prerequisite-n.test.ts', 'test/fixtures/ceo-prerequisite-n-call.json', 'test/fixtures/eng-prerequisite-77.json', 'test/fixtures/forcing-finding-seeds.ts', 'test/skill-e2e-plan-eng-multi-finding-batching.test.ts', 'test/plan-count-checkbox.test.ts', 'test/fixtures/ceo-checkbox-l.screen.txt', 'test/eng-devex-s-count.test.ts', 'test/fixtures/eng-devex-s-first-calls.json', 'test/fixtures/eng-devex-s-retry-calls.json', 'test/helpers/plan-count-file-permission.ts', 'test/plan-edit-cropped-permission.test.ts', 'test/fixtures/plan-edit-cropped-permission-1579.json', 'test/plan-count-file-permission.test.ts', 'test/fixtures/plan-count-edit-permission-t.json', 'test/eng-first-review-t.test.ts', 'test/fixtures/eng-batching-t-calls.json', 'test/plan-count-preview-footer.test.ts', 'test/fixtures/ceo-preview-u-call.json', 'test/fixtures/ceo-preview-u-screen.txt', 'test/fixtures/design-preview-v-screen.txt', 'test/plan-count-owned-permission.test.ts', 'test/fixtures/plan-count-owned-permission-v.json', 'test/fixtures/ceo-questionless-w-native.json', 'test/eng-scope-y.test.ts', 'test/fixtures/eng-scope-y-calls.json', 'test/plan-count-truncated-border.test.ts', 'test/fixtures/eng-d2-truncated-border-0bcd.json', 'test/plan-count-truncated-question.test.ts', 'test/plan-count-clipped-elision.test.ts', 'test/fixtures/eng-d1-clipped-elision-1579.json', 'test/fixtures/eng-d2-planning-prelude-4d.json', 'test/fixtures/ceo-approach-z-call.json', 'test/fixtures/ceo-approach-z-screen.txt', 'test/eng-binding-z.test.ts', 'test/fixtures/eng-binding-z-calls.json', 'test/eng-binding-retry-z.test.ts', 'test/fixtures/eng-binding-retry-z-calls.json', 'test/plan-count-permission-ac.test.ts', 'test/fixtures/plan-count-permission-ac.json', 'test/fixtures/plan-count-permission-ad.json', 'test/fixtures/plan-count-permission-ae.json', 'test/fixtures/plan-count-permission-target-ad-v2.json', 'test/eng-count-ad-v2.test.ts', 'test/fixtures/eng-count-ad-v2.json', 'test/eng-first-category-af.test.ts', 'test/fixtures/eng-first-category-af.json', 'test/fixtures/plan-count-permission-ah.json', - 'test/plan-count-crop-ak.test.ts', +"test/fixtures/batching-permission-at.json", +"test/fixtures/eng-declared-retry-at.json",'test/pty-screen-unicode-ap.test.ts','test/fixtures/design-crop-gutter-ap.json', 'test/eng-scope-entry-ap.test.ts', 'bin/gstack-config', 'bin/gstack-skill-start', 'bin/gstack-skill-end', 'plan-eng-review/**', 'scripts/resolvers/preamble.ts', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/preamble/generate-completion-status.ts', 'scripts/resolvers/review.ts', 'test/helpers/claude-pty-runner.ts', 'test/plan-count-design-ui-recovery.test.ts', 'test/fixtures/design-ui-boxed-question.json', 'test/plan-count-native-input.test.ts', 'test/helpers/pty-screen.ts', 'test/pty-screen.test.ts', 'test/pty-screen-session.test.ts', 'test/fixtures/pty-screen/**', 'test/helpers/hermetic-skill-runtime.ts', 'test/hermetic-skill-runtime.test.ts', 'test/helpers/claude-pty-runner.unit.test.ts', 'test/helpers/pty-trust-dialog.ts', 'test/pty-trust-dialog.test.ts', 'test/helpers/plan-count-transcript.ts', 'test/autoplan-public-narration.test.ts', 'test/fixtures/autoplan-public-narration-ad.json', 'test/helpers/plan-count-pending-exit.ts', 'test/plan-count-pending-exit.test.ts', 'test/plan-count-transcript.test.ts', 'test/helpers/plan-count-artifacts.ts', 'test/plan-count-artifacts.test.ts', 'test/helpers/eval-store.ts', 'test/helpers/plan-count-fixture.ts', 'test/plan-count-fixture.test.ts', 'test/plan-count-collection-completion.test.ts', 'test/plan-count-timeout.test.ts', 'test/plan-count-prerequisite.test.ts','test/fixtures/ceo-prerequisite-n-call.json', 'test/fixtures/eng-prerequisite-77.json', 'test/fixtures/forcing-finding-seeds.ts', 'test/skill-e2e-plan-eng-multi-finding-batching.test.ts', 'test/plan-count-checkbox.test.ts', 'test/fixtures/ceo-checkbox-l.screen.txt', 'test/eng-devex-s-count.test.ts', 'test/fixtures/eng-devex-s-first-calls.json', 'test/fixtures/eng-devex-s-retry-calls.json', 'test/helpers/plan-count-file-permission.ts', 'test/plan-edit-cropped-permission.test.ts', 'test/fixtures/plan-edit-cropped-permission-1579.json', 'test/plan-count-file-permission.test.ts', 'test/fixtures/plan-count-edit-permission-t.json','test/fixtures/eng-batching-t-calls.json', 'test/plan-count-preview-footer.test.ts', 'test/fixtures/ceo-preview-u-call.json', 'test/fixtures/ceo-preview-u-screen.txt', 'test/fixtures/design-preview-v-screen.txt', 'test/plan-count-owned-permission.test.ts', 'test/fixtures/plan-count-owned-permission-v.json', 'test/fixtures/ceo-questionless-w-native.json','test/fixtures/eng-scope-y-calls.json', 'test/plan-count-truncated-border.test.ts', 'test/fixtures/eng-d2-truncated-border-0bcd.json', 'test/plan-count-truncated-question.test.ts', 'test/plan-count-clipped-elision.test.ts', 'test/fixtures/eng-d1-clipped-elision-1579.json', 'test/fixtures/eng-d2-planning-prelude-4d.json', 'test/fixtures/ceo-approach-z-call.json', 'test/fixtures/ceo-approach-z-screen.txt','test/fixtures/eng-binding-z-calls.json', 'test/eng-first-review.test.ts', 'test/fixtures/eng-binding-retry-z-calls.json','test/fixtures/plan-count-permission-ac.json', 'test/fixtures/plan-count-permission-ad.json', 'test/fixtures/plan-count-permission-ae.json', 'test/fixtures/plan-count-permission-target-ad-v2.json','test/fixtures/eng-count-ad-v2.json','test/fixtures/eng-first-category-af.json', 'test/fixtures/plan-count-permission-ah.json', + 'test/fixtures/plan-count-crop-ak.json', - 'test/plan-count-quoted-frame-ak.test.ts', + 'test/fixtures/plan-count-quoted-frame-ak.json', - "test/eng-injected-export-aq.test.ts", "test/fixtures/eng-injected-export-aq.json", "test/eng-library-hooks-aq.test.ts", "test/fixtures/eng-library-hooks-aq.json", - "test/eng-declarative-as.test.ts", "test/fixtures/eng-declarative-as.json", "test/helpers/eng-cache-writer-decision.ts", "test/eng-cache-writes-as.test.ts", "test/fixtures/eng-cache-writes-as.json", +"test/fixtures/eng-injected-export-aq.json","test/fixtures/eng-library-hooks-aq.json", +"test/fixtures/eng-declarative-as.json", "test/helpers/eng-cache-writer-decision.ts","test/fixtures/eng-cache-writes-as.json", "test/review-entry-and-design-clarity-au.test.ts", "scripts/resolvers/preamble/generate-preamble-bash.ts", 'test/pty-workspace-trust.test.ts', 'test/fixtures/pty-companion-cli.ts', 'lib/fs-atomic.ts', 'test/helpers/owned-claude-transcript.ts', 'test/eval-budgets-policy.test.ts', 'test/fixtures/webfetch-permission.json', 'test/plan-skill-webfetch-permission.test.ts', 'test/helpers/plan-skill-questions.ts', 'test/fixtures/eng-auq-validation-error.json', 'test/fixtures/bash-directory-permission.json', 'test/fixtures/design-tasks-bash-permission.json', 'test/plan-skill-read-permission.test.ts', 'test/fixtures/read-permission.json', 'test/pty-numbered-option-indent-native.test.ts', 'test/fixtures/ceo-split-e5-numbered-description-491.json', 'test/plan-skill-questions.test.ts', 'test/helpers/plan-skill-question-events.ts', 'test/plan-skill-question-events.test.ts', 'test/helpers/plan-skill-question-hook-scope.ts', 'test/helpers/skill-census.ts', 'test/plan-skill-question-hook-scope.test.ts', 'test/helpers/ceo-finding-fixture.ts', 'test/ceo-finding-fixture.test.ts', 'test/helpers/plan-review-decisions.ts', 'test/plan-review-decisions.test.ts', 'test/helpers/plan-review-cases.ts', 'test/plan-review-cases.test.ts', 'test/helpers/llm-judge.ts', 'lib/eval-model.ts', 'test/skill-e2e-plan-decision-classification.test.ts', 'test/fixtures/plan-decision-classification.ts', 'test/plan-review-calibration.test.ts', 'scripts/resolvers/testing.ts', 'test/fixtures/eng-file-permission-repaint.json' @@ -545,10 +545,10 @@ export const E2E_TOUCHFILES: Record = { 'test/fixtures/plan-count-cropped-wrap-6714.json', 'test/eng-finding-retry-budget.test.ts', "test/plan-count-cross-cwd-ancestry.test.ts", "test/fixtures/plan-count-cross-cwd-ancestry-0bcd.json", "test/plan-count-session-cwd.test.ts", -'test/pty-screen-unicode-ap.test.ts', 'test/design-crop-gutter-ap.test.ts', 'test/fixtures/design-crop-gutter-ap.json', 'bin/gstack-config', 'plan-ceo-review/**', 'scripts/resolvers/preamble.ts', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'bin/gstack-question-preference', 'test/helpers/claude-pty-runner.ts', 'test/plan-count-design-ui-recovery.test.ts', 'test/fixtures/design-ui-boxed-question.json', 'test/plan-count-native-input.test.ts', 'test/helpers/pty-screen.ts', 'test/pty-screen.test.ts', 'test/pty-screen-session.test.ts', 'test/fixtures/pty-screen/**', 'test/helpers/hermetic-skill-runtime.ts', 'test/hermetic-skill-runtime.test.ts', 'test/helpers/claude-pty-runner.unit.test.ts', 'test/helpers/pty-trust-dialog.ts', 'test/pty-trust-dialog.test.ts', 'test/helpers/plan-count-transcript.ts', 'test/autoplan-public-narration.test.ts', 'test/fixtures/autoplan-public-narration-ad.json', 'test/helpers/plan-count-pending-exit.ts', 'test/plan-count-pending-exit.test.ts', 'test/plan-count-transcript.test.ts', 'test/helpers/plan-count-artifacts.ts', 'test/plan-count-artifacts.test.ts', 'test/helpers/eval-store.ts', 'test/helpers/plan-count-fixture.ts', 'test/plan-count-fixture.test.ts', 'test/plan-count-collection-completion.test.ts', 'test/plan-count-timeout.test.ts', 'test/plan-count-navigation-r.test.ts', 'test/plan-count-prerequisite-n.test.ts', 'test/fixtures/ceo-prerequisite-n-call.json', 'test/fixtures/eng-prerequisite-77.json', 'test/fixtures/forcing-finding-seeds.ts', 'test/skill-e2e-plan-ceo-split-overflow.test.ts', 'test/plan-count-checkbox.test.ts', 'test/fixtures/ceo-checkbox-l.screen.txt', 'test/helpers/plan-count-file-permission.ts', 'test/plan-edit-cropped-permission.test.ts', 'test/fixtures/plan-edit-cropped-permission-1579.json', 'test/plan-count-file-permission.test.ts', 'test/fixtures/plan-count-edit-permission-t.json', 'test/plan-count-owned-permission.test.ts', 'test/fixtures/plan-count-owned-permission-v.json', 'test/fixtures/ceo-questionless-w-native.json', 'test/plan-count-truncated-border.test.ts', 'test/fixtures/eng-d2-truncated-border-0bcd.json', 'test/plan-count-truncated-question.test.ts', 'test/plan-count-clipped-elision.test.ts', 'test/fixtures/eng-d1-clipped-elision-1579.json', 'test/fixtures/eng-d2-planning-prelude-4d.json', 'test/fixtures/ceo-approach-z-call.json', 'test/fixtures/ceo-approach-z-screen.txt', 'test/plan-count-permission-ac.test.ts', 'test/fixtures/plan-count-permission-ac.json', 'test/fixtures/plan-count-permission-ad.json', 'test/fixtures/plan-count-permission-ae.json', 'test/fixtures/plan-count-permission-target-ad-v2.json', 'test/fixtures/plan-count-permission-ah.json', - 'test/plan-count-crop-ak.test.ts', +'test/pty-screen-unicode-ap.test.ts','test/fixtures/design-crop-gutter-ap.json', 'bin/gstack-config', 'plan-ceo-review/**', 'scripts/resolvers/preamble.ts', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'bin/gstack-question-preference', 'test/helpers/claude-pty-runner.ts', 'test/plan-count-design-ui-recovery.test.ts', 'test/fixtures/design-ui-boxed-question.json', 'test/plan-count-native-input.test.ts', 'test/helpers/pty-screen.ts', 'test/pty-screen.test.ts', 'test/pty-screen-session.test.ts', 'test/fixtures/pty-screen/**', 'test/helpers/hermetic-skill-runtime.ts', 'test/hermetic-skill-runtime.test.ts', 'test/helpers/claude-pty-runner.unit.test.ts', 'test/helpers/pty-trust-dialog.ts', 'test/pty-trust-dialog.test.ts', 'test/helpers/plan-count-transcript.ts', 'test/autoplan-public-narration.test.ts', 'test/fixtures/autoplan-public-narration-ad.json', 'test/helpers/plan-count-pending-exit.ts', 'test/plan-count-pending-exit.test.ts', 'test/plan-count-transcript.test.ts', 'test/helpers/plan-count-artifacts.ts', 'test/plan-count-artifacts.test.ts', 'test/helpers/eval-store.ts', 'test/helpers/plan-count-fixture.ts', 'test/plan-count-fixture.test.ts', 'test/plan-count-collection-completion.test.ts', 'test/plan-count-timeout.test.ts', 'test/plan-count-prerequisite.test.ts','test/fixtures/ceo-prerequisite-n-call.json', 'test/fixtures/eng-prerequisite-77.json', 'test/fixtures/forcing-finding-seeds.ts', 'test/skill-e2e-plan-ceo-split-overflow.test.ts', 'test/plan-count-checkbox.test.ts', 'test/fixtures/ceo-checkbox-l.screen.txt', 'test/helpers/plan-count-file-permission.ts', 'test/plan-edit-cropped-permission.test.ts', 'test/fixtures/plan-edit-cropped-permission-1579.json', 'test/plan-count-file-permission.test.ts', 'test/fixtures/plan-count-edit-permission-t.json', 'test/plan-count-owned-permission.test.ts', 'test/fixtures/plan-count-owned-permission-v.json', 'test/fixtures/ceo-questionless-w-native.json', 'test/plan-count-truncated-border.test.ts', 'test/fixtures/eng-d2-truncated-border-0bcd.json', 'test/plan-count-truncated-question.test.ts', 'test/plan-count-clipped-elision.test.ts', 'test/fixtures/eng-d1-clipped-elision-1579.json', 'test/fixtures/eng-d2-planning-prelude-4d.json', 'test/fixtures/ceo-approach-z-call.json', 'test/fixtures/ceo-approach-z-screen.txt','test/fixtures/plan-count-permission-ac.json', 'test/fixtures/plan-count-permission-ad.json', 'test/fixtures/plan-count-permission-ae.json', 'test/fixtures/plan-count-permission-target-ad-v2.json', 'test/fixtures/plan-count-permission-ah.json', + 'test/fixtures/plan-count-crop-ak.json', - 'test/plan-count-quoted-frame-ak.test.ts', + 'test/fixtures/plan-count-quoted-frame-ak.json', 'docs/askuserquestion-split.md', 'test/resolver-ask-user-format.test.ts', 'bin/gstack-slug', 'test/helpers/hermetic-env.test.ts', 'test/pty-workspace-trust.test.ts', 'test/fixtures/pty-companion-cli.ts', 'lib/fs-atomic.ts', 'test/helpers/ceo-finding-fixture.ts', 'test/ceo-finding-fixture.test.ts', 'test/helpers/owned-claude-transcript.ts', 'test/eval-budgets-policy.test.ts', 'test/fixtures/webfetch-permission.json', 'test/plan-skill-webfetch-permission.test.ts', 'test/helpers/plan-skill-questions.ts', 'test/fixtures/eng-auq-validation-error.json', 'test/fixtures/bash-directory-permission.json', 'test/fixtures/design-tasks-bash-permission.json', 'test/plan-skill-read-permission.test.ts', 'test/fixtures/read-permission.json', 'test/pty-numbered-option-indent-native.test.ts', 'test/fixtures/ceo-split-e5-numbered-description-491.json', 'test/plan-skill-questions.test.ts', 'test/helpers/plan-skill-question-events.ts', 'test/plan-skill-question-events.test.ts', 'test/helpers/plan-skill-question-hook-scope.ts', 'test/helpers/skill-census.ts', 'test/plan-skill-question-hook-scope.test.ts', 'test/helpers/plan-review-decisions.ts', 'test/plan-review-decisions.test.ts', 'test/helpers/plan-review-cases.ts', 'test/plan-review-cases.test.ts', 'test/helpers/llm-judge.ts', 'lib/eval-model.ts', 'test/skill-e2e-plan-decision-classification.test.ts', 'test/fixtures/plan-decision-classification.ts', 'test/plan-review-calibration.test.ts', 'test/helpers/ceo-split-question-policy.ts', 'test/fixtures/ceo-split-actor-6aef.json', 'test/helpers/ceo-mode-option.ts', 'test/ceo-mode-option.test.ts', 'test/ceo-split-collection.test.ts', 'test/fixtures/ceo-split-collection-0bcd.json', 'test/ceo-split-question-policy.test.ts', 'scripts/resolvers/review.ts', 'scripts/resolvers/tasks-section.ts' ], @@ -582,14 +582,14 @@ export const E2E_TOUCHFILES: Record = { ], 'plan-eng-review-format-coverage': ['test/session-runner-stream-lifecycle.test.ts', 'test/paid-retry-supervision.test.ts', 'scripts/resolvers/learnings.ts', - "test/plan-scope-recovery-av.test.ts", + "test/plan-scope-selection.test.ts", "test/fixtures/plan-scope-recovery-av.json", 'test/eng-scope-entry-ap.test.ts', 'plan-eng-review/**', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/preamble/generate-completeness-section.ts', 'scripts/resolvers/preamble.ts', 'model-overlays/opus-4-7.md', 'test/helpers/llm-judge.ts', 'test/skill-e2e-plan-format.test.ts', "test/review-entry-and-design-clarity-au.test.ts", "scripts/resolvers/preamble/generate-preamble-bash.ts", "scripts/resolvers/preamble/generate-completion-status.ts", 'scripts/resolvers/testing.ts', 'test/helpers/office-hours-attempt.ts', 'test/office-hours-attempt.test.ts', 'scripts/resolvers/review.ts', 'test/plan-review-cases.test.ts' ], 'plan-eng-review-format-kind': ['test/session-runner-stream-lifecycle.test.ts', 'test/paid-retry-supervision.test.ts', 'scripts/resolvers/learnings.ts', - "test/plan-scope-recovery-av.test.ts", + "test/plan-scope-selection.test.ts", "test/fixtures/plan-scope-recovery-av.json", 'test/eng-scope-entry-ap.test.ts', 'plan-eng-review/**', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/preamble/generate-completeness-section.ts', 'scripts/resolvers/preamble.ts', 'model-overlays/opus-4-7.md', 'test/helpers/llm-judge.ts', 'test/skill-e2e-plan-format.test.ts', "test/review-entry-and-design-clarity-au.test.ts", "scripts/resolvers/preamble/generate-preamble-bash.ts", "scripts/resolvers/preamble/generate-completion-status.ts", 'scripts/resolvers/testing.ts', 'test/helpers/office-hours-attempt.ts', 'test/office-hours-attempt.test.ts', 'scripts/resolvers/review.ts', 'test/plan-review-cases.test.ts' @@ -599,19 +599,19 @@ export const E2E_TOUCHFILES: Record = { // Dependencies: same as format-mode + the 4 plan-review templates + overlay. // All periodic-tier (non-deterministic Opus 4.7 behavior). 'plan-ceo-review-prosons-cadence': ['test/session-runner-stream-lifecycle.test.ts', 'test/paid-retry-supervision.test.ts', - "test/plan-scope-recovery-av.test.ts", + "test/fixtures/plan-scope-recovery-av.json", "test/fixtures/design-scope-checkpoint-at.json",'test/eng-scope-entry-ap.test.ts', 'plan-ceo-review/**', 'plan-eng-review/**', 'plan-design-review/**', 'plan-devex-review/**', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/preamble.ts', 'model-overlays/opus-4-7.md', 'test/skill-e2e-plan-prosons.test.ts', - "test/design-scope-entry-aq.test.ts", + "test/plan-scope-selection.test.ts", "test/review-entry-and-design-clarity-au.test.ts", "scripts/resolvers/preamble/generate-preamble-bash.ts", "scripts/resolvers/preamble/generate-completion-status.ts", 'scripts/resolvers/testing.ts', 'test/helpers/office-hours-attempt.ts', 'test/office-hours-attempt.test.ts', 'scripts/resolvers/review.ts', 'test/plan-review-cases.test.ts', 'scripts/resolvers/tasks-section.ts' ], 'plan-review-prosons-format': ['test/session-runner-stream-lifecycle.test.ts', 'test/paid-retry-supervision.test.ts', - "test/plan-scope-recovery-av.test.ts", + "test/fixtures/plan-scope-recovery-av.json", "test/fixtures/design-scope-checkpoint-at.json",'test/eng-scope-entry-ap.test.ts', 'plan-ceo-review/**', 'plan-eng-review/**', 'plan-design-review/**', 'plan-devex-review/**', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/preamble.ts', 'model-overlays/opus-4-7.md', 'test/skill-e2e-plan-prosons.test.ts', - "test/design-scope-entry-aq.test.ts", + "test/plan-scope-selection.test.ts", "test/review-entry-and-design-clarity-au.test.ts", "scripts/resolvers/preamble/generate-preamble-bash.ts", "scripts/resolvers/preamble/generate-completion-status.ts", 'scripts/resolvers/testing.ts', 'test/helpers/office-hours-attempt.ts', 'test/office-hours-attempt.test.ts', 'scripts/resolvers/review.ts', 'test/plan-review-cases.test.ts', 'scripts/resolvers/tasks-section.ts' @@ -643,10 +643,10 @@ export const E2E_TOUCHFILES: Record = { 'codex-offered-design-review': ['test/session-runner-stream-lifecycle.test.ts', 'test/paid-retry-supervision.test.ts', 'test/helpers/codex-offering-fixture.ts', 'test/codex-offering-fixture.test.ts', 'test/fixtures/codex-offering-cdd-public.json', 'test/fixtures/codex-offering-timeout-public.json', 'test/helpers/workflow-judge-input.ts', 'test/workflow-judge-input.test.ts', 'test/helpers/workflow-excerpt.ts', - "test/plan-scope-recovery-av.test.ts", + "test/fixtures/plan-scope-recovery-av.json", "test/fixtures/design-scope-checkpoint-at.json",'plan-design-review/**', 'scripts/gen-skill-docs.ts', 'test/skill-e2e-plan.test.ts', - "test/design-scope-entry-aq.test.ts", + "test/plan-scope-selection.test.ts", "test/review-entry-and-design-clarity-au.test.ts", "scripts/resolvers/preamble/generate-preamble-bash.ts", "scripts/resolvers/preamble/generate-completion-status.ts", 'scripts/resolvers/preamble/generate-ask-user-format.ts' @@ -655,7 +655,7 @@ export const E2E_TOUCHFILES: Record = { 'test/helpers/codex-offering-fixture.ts', 'test/codex-offering-fixture.test.ts', 'test/fixtures/codex-offering-cdd-public.json', 'test/fixtures/codex-offering-timeout-public.json', 'test/helpers/workflow-judge-input.ts', 'test/workflow-judge-input.test.ts', 'test/helpers/workflow-excerpt.ts', 'scripts/resolvers/learnings.ts', - "test/plan-scope-recovery-av.test.ts", + "test/plan-scope-selection.test.ts", "test/fixtures/plan-scope-recovery-av.json", 'test/eng-scope-entry-ap.test.ts', 'plan-eng-review/**', 'scripts/gen-skill-docs.ts', 'test/skill-e2e-plan.test.ts', "test/review-entry-and-design-clarity-au.test.ts", "scripts/resolvers/preamble/generate-preamble-bash.ts", "scripts/resolvers/preamble/generate-completion-status.ts", 'scripts/resolvers/testing.ts', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/review.ts', 'test/plan-review-cases.test.ts' @@ -735,7 +735,7 @@ export const E2E_TOUCHFILES: Record = { 'scripts/gen-skill-docs.ts', 'scripts/resolvers/index.ts', 'scripts/resolvers/outside-voice.ts', 'scripts/resolvers/constants.ts', 'scripts/resolvers/review.ts', 'bin/gstack-claude-code', 'lib/claude-code.ts', 'lib/claude-code-windows-job.ts', 'lib/claude-bin.ts', 'lib/outside-review-result.ts', 'test/helpers/codex-session-runner.ts', - 'test/helpers/skill-fixture.ts', 'test/helpers/outside-voice-fixture.ts', 'test/outside-voice-fixture.test.ts', 'test/helpers/outside-voice-evidence.ts', 'test/outside-voice-evidence.test.ts', 'test/outside-voice-async.test.ts', 'test/fixtures/outside-async-task-m-events.json', 'test/skill-e2e-outside-voice.test.ts', + 'test/helpers/skill-fixture.ts', 'test/helpers/outside-voice-fixture.ts', 'test/outside-voice-fixture.test.ts', 'test/helpers/outside-voice-evidence.ts', 'test/outside-voice-evidence.test.ts','test/fixtures/outside-async-task-m-events.json', 'test/skill-e2e-outside-voice.test.ts', 'test/helpers/outside-voice-receipt.ts', 'test/outside-voice-receipt.test.ts', 'test/claude-code-runner.test.ts', ], @@ -744,15 +744,15 @@ export const E2E_TOUCHFILES: Record = { 'scripts/gen-skill-docs.ts', 'scripts/resolvers/index.ts', 'scripts/resolvers/outside-voice.ts', 'scripts/resolvers/constants.ts', 'scripts/resolvers/review.ts', 'bin/gstack-codex-probe', 'lib/outside-review-result.ts', 'test/helpers/session-runner.ts', - 'test/outside-background-ai.test.ts', 'test/fixtures/outside-background-ai.json', - 'test/helpers/skill-fixture.ts', 'test/helpers/outside-voice-fixture.ts', 'test/outside-voice-fixture.test.ts', 'test/helpers/outside-voice-evidence.ts', 'test/outside-voice-evidence.test.ts', 'test/outside-voice-async.test.ts', 'test/fixtures/outside-async-task-m-events.json', 'test/skill-e2e-outside-voice.test.ts', +'test/fixtures/outside-background-ai.json', + 'test/helpers/skill-fixture.ts', 'test/helpers/outside-voice-fixture.ts', 'test/outside-voice-fixture.test.ts', 'test/helpers/outside-voice-evidence.ts', 'test/outside-voice-evidence.test.ts','test/fixtures/outside-async-task-m-events.json', 'test/skill-e2e-outside-voice.test.ts', ], // Disabled means no extra plan review, including a native Agent fallback. 'outside-plan-disabled-no-fallback': ['test/session-runner-stream-lifecycle.test.ts', 'test/fixtures/disabled-retained-record.json', 'scripts/resolvers/testing.ts', 'scripts/resolvers/preamble/generate-ask-user-format.ts', - "test/plan-scope-recovery-av.test.ts", + "test/plan-scope-selection.test.ts", "test/fixtures/plan-scope-recovery-av.json", "test/disabled-dated-record-at.test.ts", "test/fixtures/disabled-dated-record-at.json",'test/eng-scope-entry-ap.test.ts', 'plan-eng-review/**', 'plan-ceo-review/**', 'hosts/claude.ts', 'hosts/define-host.ts', @@ -777,42 +777,42 @@ export const E2E_TOUCHFILES: Record = { // Coverage audit (shared fixture) + triage + gates 'ship-coverage-audit': ['test/session-runner-stream-lifecycle.test.ts', - "test/coverage-audit-aw.test.ts", + "test/coverage-audit-evidence.test.ts", "test/fixtures/coverage-audit-aw.json", - "test/coverage-checkbox-tail-av.test.ts", + "test/fixtures/coverage-checkbox-tail-av.json", 'test/helpers/coverage-audit-evidence.ts', 'test/ship-coverage-audit-af.test.ts', 'test/fixtures/ship-coverage-audit-af.json','ship/**', 'test/fixtures/coverage-audit-fixture.ts', 'bin/gstack-repo-mode', 'test/skill-e2e-workflow.test.ts', - "test/coverage-shell-display-aq.test.ts", "test/fixtures/coverage-shell-display-aq.json", +"test/fixtures/coverage-shell-display-aq.json", 'test/helpers/coverage-audit.ts', 'test/helpers/office-hours-attempt.ts', 'test/coverage-audit.test.ts', 'scripts/resolvers/testing.ts' ], 'review-coverage-audit': ['test/session-runner-stream-lifecycle.test.ts', 'test/fixtures/coverage-audit-ci-diagrams.json', - "test/coverage-audit-aw.test.ts", + "test/fixtures/coverage-audit-aw.json", - "test/coverage-checkbox-tail-av.test.ts", + "test/fixtures/coverage-checkbox-tail-av.json", - "test/coverage-audit-shell-legend-at.test.ts", "test/fixtures/coverage-audit-shell-legend-at.json",'review/**', 'test/fixtures/coverage-audit-fixture.ts', 'test/skill-e2e-coverage-audit.test.ts', 'test/helpers/coverage-audit-evidence.ts', 'test/coverage-audit-evidence.test.ts', 'test/fixtures/coverage-audit-ae.json', 'test/coverage-audit-af.test.ts', 'test/fixtures/coverage-audit-af.json', - "test/coverage-shell-display-aq.test.ts", "test/fixtures/coverage-shell-display-aq.json", - "test/coverage-diagram-legend-as.test.ts", "test/fixtures/coverage-diagram-legend-as.json", +"test/fixtures/coverage-audit-shell-legend-at.json",'review/**', 'test/fixtures/coverage-audit-fixture.ts', 'test/skill-e2e-coverage-audit.test.ts', 'test/helpers/coverage-audit-evidence.ts', 'test/coverage-audit-evidence.test.ts', 'test/fixtures/coverage-audit-ae.json','test/fixtures/coverage-audit-af.json', +"test/fixtures/coverage-shell-display-aq.json", +"test/fixtures/coverage-diagram-legend-as.json", 'test/helpers/coverage-audit.ts', 'test/helpers/office-hours-attempt.ts', 'test/coverage-audit.test.ts' ], 'plan-eng-coverage-audit': ['test/session-runner-stream-lifecycle.test.ts', 'scripts/resolvers/learnings.ts', 'test/fixtures/coverage-audit-ci-diagrams.json', - "test/coverage-audit-aw.test.ts", + "test/fixtures/coverage-audit-aw.json", - "test/coverage-checkbox-tail-av.test.ts", + "test/fixtures/coverage-checkbox-tail-av.json", - "test/plan-scope-recovery-av.test.ts", + "test/plan-scope-selection.test.ts", "test/fixtures/plan-scope-recovery-av.json", - "test/coverage-audit-shell-legend-at.test.ts", "test/fixtures/coverage-audit-shell-legend-at.json",'test/eng-scope-entry-ap.test.ts', 'plan-eng-review/**', 'test/fixtures/coverage-audit-fixture.ts', 'test/skill-e2e-coverage-audit.test.ts', 'test/helpers/coverage-audit-evidence.ts', 'test/coverage-audit-evidence.test.ts', 'test/fixtures/coverage-audit-ae.json', 'test/coverage-audit-af.test.ts', 'test/fixtures/coverage-audit-af.json', - "test/coverage-shell-display-aq.test.ts", "test/fixtures/coverage-shell-display-aq.json", - "test/coverage-diagram-legend-as.test.ts", "test/fixtures/coverage-diagram-legend-as.json", +"test/fixtures/coverage-audit-shell-legend-at.json",'test/eng-scope-entry-ap.test.ts', 'plan-eng-review/**', 'test/fixtures/coverage-audit-fixture.ts', 'test/skill-e2e-coverage-audit.test.ts', 'test/helpers/coverage-audit-evidence.ts', 'test/coverage-audit-evidence.test.ts', 'test/fixtures/coverage-audit-ae.json','test/fixtures/coverage-audit-af.json', +"test/fixtures/coverage-shell-display-aq.json", +"test/fixtures/coverage-diagram-legend-as.json", "test/review-entry-and-design-clarity-au.test.ts", "scripts/resolvers/preamble/generate-preamble-bash.ts", "scripts/resolvers/preamble/generate-completion-status.ts", 'test/helpers/coverage-audit.ts', 'test/helpers/office-hours-attempt.ts', 'test/coverage-audit.test.ts', 'scripts/resolvers/testing.ts', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/review.ts', 'test/plan-review-cases.test.ts' @@ -845,10 +845,10 @@ export const E2E_TOUCHFILES: Record = { 'design-consultation-research': ['test/session-runner-stream-lifecycle.test.ts', 'design-consultation/**', 'scripts/resolvers/aside.ts', 'scripts/gen-skill-docs.ts', 'test/skill-e2e-design.test.ts', 'test/helpers/skill-fixture.ts', 'test/design-research-fixture.test.ts', 'scripts/resolvers/design.ts', 'scripts/resolvers/outside-voice.ts', 'design-consultation/sections/**', 'test/design-consultation-contract.test.ts'], 'design-consultation-preview': ['test/session-runner-stream-lifecycle.test.ts', 'design-consultation/**', 'scripts/gen-skill-docs.ts', 'test/skill-e2e-design.test.ts', 'test/design-board-reload.test.ts'], 'plan-design-review-no-ui-scope': ['test/session-runner-stream-lifecycle.test.ts', - "test/plan-scope-recovery-av.test.ts", + "test/fixtures/plan-scope-recovery-av.json", "test/fixtures/design-scope-checkpoint-at.json",'plan-design-review/**', 'lib/design-catalog.ts', 'scripts/gen-skill-docs.ts', 'test/skill-e2e-design.test.ts', - "test/design-scope-entry-aq.test.ts", + "test/plan-scope-selection.test.ts", "test/review-entry-and-design-clarity-au.test.ts", "scripts/resolvers/preamble/generate-preamble-bash.ts", "scripts/resolvers/preamble/generate-completion-status.ts", 'scripts/resolvers/preamble/generate-ask-user-format.ts' @@ -894,10 +894,10 @@ export const E2E_TOUCHFILES: Record = { 'test/fixtures/autoplan-dual-false-positive-6bd.json', 'test/helpers/autoplan-method-read-audit.ts', 'test/autoplan-method-read-audit.test.ts', 'test/fixtures/autoplan-method-read-aa-events.json', - 'test/helpers/outside-voice-evidence.ts', 'test/outside-voice-evidence.test.ts', 'test/outside-voice-async.test.ts', + 'test/helpers/outside-voice-evidence.ts', 'test/outside-voice-evidence.test.ts', 'test/fixtures/outside-async-task-m-events.json', 'test/autoplan-amend-input.test.ts', 'test/fixtures/autoplan-amend-input-77.json', - 'test/autoplan-phase-handoff.test.ts', 'test/fixtures/autoplan-phase-handoff-6714.json','scripts/resolvers/learnings.ts', 'scripts/resolvers/preamble/generate-completion-status.ts', 'scripts/resolvers/preamble/generate-preamble-bash.ts', 'test/review-entry-and-design-clarity-au.test.ts', 'test/fixtures/plan-scope-recovery-av.json', 'test/plan-scope-recovery-av.test.ts', 'test/eng-scope-entry-ap.test.ts', 'test/fixtures/design-scope-checkpoint-at.json', 'test/design-scope-entry-aq.test.ts', 'scripts/resolvers/composition.ts', 'test/autoplan-review-discovery.test.ts', 'test/autoplan-phase-order.test.ts', 'autoplan/**', 'codex/**', 'bin/gstack-codex-probe', 'scripts/resolvers/review.ts', 'scripts/resolvers/design.ts', 'test/skill-e2e-autoplan-dual-voice.test.ts', 'bin/gstack-autoplan-snapshot.ts', 'test/autoplan-snapshot.test.ts', 'test/autoplan-init.test.ts', 'test/autoplan-obligations.test.ts', 'test/fixtures/autoplan/t-ceo-omitted-obligations.json', 'test/fixtures/autoplan/u-ceo-original-loss.json', 'test/fixtures/autoplan/v-ceo-dangling-references.json', + 'test/autoplan-phase-handoff.test.ts', 'test/fixtures/autoplan-phase-handoff-6714.json','scripts/resolvers/learnings.ts', 'scripts/resolvers/preamble/generate-completion-status.ts', 'scripts/resolvers/preamble/generate-preamble-bash.ts', 'test/review-entry-and-design-clarity-au.test.ts', 'test/fixtures/plan-scope-recovery-av.json','test/eng-scope-entry-ap.test.ts', 'test/fixtures/design-scope-checkpoint-at.json', 'test/plan-scope-selection.test.ts', 'scripts/resolvers/composition.ts', 'test/autoplan-review-discovery.test.ts', 'test/autoplan-phase-order.test.ts', 'autoplan/**', 'codex/**', 'bin/gstack-codex-probe', 'scripts/resolvers/review.ts', 'scripts/resolvers/design.ts', 'test/skill-e2e-autoplan-dual-voice.test.ts', 'bin/gstack-autoplan-snapshot.ts', 'test/autoplan-snapshot.test.ts', 'test/autoplan-init.test.ts', 'test/autoplan-obligations.test.ts', 'test/fixtures/autoplan/t-ceo-omitted-obligations.json', 'test/fixtures/autoplan/u-ceo-original-loss.json', 'test/fixtures/autoplan/v-ceo-dangling-references.json', 'scripts/resolvers/design-doc-discovery.ts', 'plan-ceo-review/**', 'plan-eng-review/**', 'plan-design-review/**', 'plan-devex-review/**', 'scripts/resolvers/testing.ts', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'test/plan-review-cases.test.ts', 'scripts/resolvers/tasks-section.ts' ], @@ -935,100 +935,100 @@ export const E2E_TOUCHFILES: Record = { // Skill routing — journey-stage tests (depend on ALL skill descriptions) 'journey-ideation': ['test/session-runner-stream-lifecycle.test.ts', 'test/skill-fixture.test.ts', - "test/plan-scope-recovery-av.test.ts", + "test/fixtures/plan-scope-recovery-av.json", "test/fixtures/design-scope-checkpoint-at.json",'test/eng-scope-entry-ap.test.ts', '*/SKILL.md.tmpl', 'SKILL.md.tmpl', 'scripts/gen-skill-docs.ts', 'test/skill-routing-e2e.test.ts', - "test/design-scope-entry-aq.test.ts", + "test/plan-scope-selection.test.ts", "test/review-entry-and-design-clarity-au.test.ts", "scripts/resolvers/preamble/generate-preamble-bash.ts", "scripts/resolvers/preamble/generate-completion-status.ts", ], 'journey-plan-eng': ['test/session-runner-stream-lifecycle.test.ts', 'test/skill-fixture.test.ts', - "test/plan-scope-recovery-av.test.ts", + "test/fixtures/plan-scope-recovery-av.json", "test/fixtures/design-scope-checkpoint-at.json",'test/eng-scope-entry-ap.test.ts', '*/SKILL.md.tmpl', 'SKILL.md.tmpl', 'scripts/gen-skill-docs.ts', 'test/skill-routing-e2e.test.ts', - "test/design-scope-entry-aq.test.ts", + "test/plan-scope-selection.test.ts", "test/review-entry-and-design-clarity-au.test.ts", "scripts/resolvers/preamble/generate-preamble-bash.ts", "scripts/resolvers/preamble/generate-completion-status.ts", ], 'journey-debug': ['test/session-runner-stream-lifecycle.test.ts', 'test/skill-fixture.test.ts', - "test/plan-scope-recovery-av.test.ts", + "test/fixtures/plan-scope-recovery-av.json", "test/fixtures/design-scope-checkpoint-at.json",'test/eng-scope-entry-ap.test.ts', '*/SKILL.md.tmpl', 'SKILL.md.tmpl', 'scripts/gen-skill-docs.ts', 'test/skill-routing-e2e.test.ts', - "test/design-scope-entry-aq.test.ts", + "test/plan-scope-selection.test.ts", "test/review-entry-and-design-clarity-au.test.ts", "scripts/resolvers/preamble/generate-preamble-bash.ts", "scripts/resolvers/preamble/generate-completion-status.ts", ], 'journey-qa': ['test/session-runner-stream-lifecycle.test.ts', 'test/skill-fixture.test.ts', - "test/plan-scope-recovery-av.test.ts", + "test/fixtures/plan-scope-recovery-av.json", "test/fixtures/design-scope-checkpoint-at.json",'test/eng-scope-entry-ap.test.ts', '*/SKILL.md.tmpl', 'SKILL.md.tmpl', 'scripts/gen-skill-docs.ts', 'test/skill-routing-e2e.test.ts', - "test/design-scope-entry-aq.test.ts", + "test/plan-scope-selection.test.ts", "test/review-entry-and-design-clarity-au.test.ts", "scripts/resolvers/preamble/generate-preamble-bash.ts", "scripts/resolvers/preamble/generate-completion-status.ts", ], 'journey-code-review': ['test/session-runner-stream-lifecycle.test.ts', 'test/skill-fixture.test.ts', - "test/plan-scope-recovery-av.test.ts", + "test/fixtures/plan-scope-recovery-av.json", "test/fixtures/design-scope-checkpoint-at.json",'test/eng-scope-entry-ap.test.ts', '*/SKILL.md.tmpl', 'SKILL.md.tmpl', 'scripts/gen-skill-docs.ts', 'test/skill-routing-e2e.test.ts', - "test/design-scope-entry-aq.test.ts", + "test/plan-scope-selection.test.ts", "test/review-entry-and-design-clarity-au.test.ts", "scripts/resolvers/preamble/generate-preamble-bash.ts", "scripts/resolvers/preamble/generate-completion-status.ts", ], 'journey-ship': ['test/session-runner-stream-lifecycle.test.ts', 'test/skill-fixture.test.ts', - "test/plan-scope-recovery-av.test.ts", + "test/fixtures/plan-scope-recovery-av.json", "test/fixtures/design-scope-checkpoint-at.json",'test/eng-scope-entry-ap.test.ts', '*/SKILL.md.tmpl', 'SKILL.md.tmpl', 'scripts/gen-skill-docs.ts', 'test/skill-routing-e2e.test.ts', - "test/design-scope-entry-aq.test.ts", + "test/plan-scope-selection.test.ts", "test/review-entry-and-design-clarity-au.test.ts", "scripts/resolvers/preamble/generate-preamble-bash.ts", "scripts/resolvers/preamble/generate-completion-status.ts", ], 'journey-docs': ['test/session-runner-stream-lifecycle.test.ts', 'test/skill-fixture.test.ts', - "test/plan-scope-recovery-av.test.ts", + "test/fixtures/plan-scope-recovery-av.json", "test/fixtures/design-scope-checkpoint-at.json",'test/eng-scope-entry-ap.test.ts', '*/SKILL.md.tmpl', 'SKILL.md.tmpl', 'scripts/gen-skill-docs.ts', 'test/skill-routing-e2e.test.ts', - "test/design-scope-entry-aq.test.ts", + "test/plan-scope-selection.test.ts", "test/review-entry-and-design-clarity-au.test.ts", "scripts/resolvers/preamble/generate-preamble-bash.ts", "scripts/resolvers/preamble/generate-completion-status.ts", ], 'journey-retro': ['test/session-runner-stream-lifecycle.test.ts', 'test/skill-fixture.test.ts', - "test/plan-scope-recovery-av.test.ts", + "test/fixtures/plan-scope-recovery-av.json", "test/fixtures/design-scope-checkpoint-at.json",'test/eng-scope-entry-ap.test.ts', '*/SKILL.md.tmpl', 'SKILL.md.tmpl', 'scripts/gen-skill-docs.ts', 'test/skill-routing-e2e.test.ts', - "test/design-scope-entry-aq.test.ts", + "test/plan-scope-selection.test.ts", "test/review-entry-and-design-clarity-au.test.ts", "scripts/resolvers/preamble/generate-preamble-bash.ts", "scripts/resolvers/preamble/generate-completion-status.ts", ], 'journey-design-system': ['test/session-runner-stream-lifecycle.test.ts', 'test/skill-fixture.test.ts', - "test/plan-scope-recovery-av.test.ts", + "test/fixtures/plan-scope-recovery-av.json", "test/fixtures/design-scope-checkpoint-at.json",'test/eng-scope-entry-ap.test.ts', '*/SKILL.md.tmpl', 'SKILL.md.tmpl', 'scripts/gen-skill-docs.ts', 'test/skill-routing-e2e.test.ts', - "test/design-scope-entry-aq.test.ts", + "test/plan-scope-selection.test.ts", "test/review-entry-and-design-clarity-au.test.ts", "scripts/resolvers/preamble/generate-preamble-bash.ts", "scripts/resolvers/preamble/generate-completion-status.ts", ], 'journey-visual-qa': ['test/session-runner-stream-lifecycle.test.ts', 'test/skill-fixture.test.ts', - "test/plan-scope-recovery-av.test.ts", + "test/fixtures/plan-scope-recovery-av.json", "test/fixtures/design-scope-checkpoint-at.json",'test/eng-scope-entry-ap.test.ts', '*/SKILL.md.tmpl', 'SKILL.md.tmpl', 'scripts/gen-skill-docs.ts', 'test/skill-routing-e2e.test.ts', - "test/design-scope-entry-aq.test.ts", + "test/plan-scope-selection.test.ts", "test/review-entry-and-design-clarity-au.test.ts", "scripts/resolvers/preamble/generate-preamble-bash.ts", "scripts/resolvers/preamble/generate-completion-status.ts", ], 'journey-negatives': ['test/session-runner-stream-lifecycle.test.ts', 'test/skill-fixture.test.ts', - "test/plan-scope-recovery-av.test.ts", + "test/fixtures/plan-scope-recovery-av.json", "test/fixtures/design-scope-checkpoint-at.json",'test/eng-scope-entry-ap.test.ts', '*/SKILL.md.tmpl', 'SKILL.md.tmpl', 'scripts/gen-skill-docs.ts', 'test/skill-routing-e2e.test.ts', - "test/design-scope-entry-aq.test.ts", + "test/plan-scope-selection.test.ts", "test/review-entry-and-design-clarity-au.test.ts", "scripts/resolvers/preamble/generate-preamble-bash.ts", "scripts/resolvers/preamble/generate-completion-status.ts", ], @@ -1493,7 +1493,7 @@ export const LLM_JUDGE_TOUCHFILES: Record = { 'plan-eng-review/SKILL.md sections': [ 'test/eng-review-routing.test.ts', 'scripts/resolvers/gbrain.ts', 'scripts/resolvers/learnings.ts', - "test/plan-scope-recovery-av.test.ts", + "test/plan-scope-selection.test.ts", "test/fixtures/plan-scope-recovery-av.json", 'test/eng-scope-entry-ap.test.ts', 'plan-eng-review/SKILL.md', 'plan-eng-review/SKILL.md.tmpl', 'test/skill-llm-eval.test.ts', 'test/helpers/workflow-judge-input.ts', 'test/helpers/workflow-judge-cache.ts', 'test/workflow-judge-cache.test.ts', 'scripts/eval-input-cache.ts', 'test/eval-input-cache.test.ts', 'test/workflow-judge-input.test.ts', 'test/helpers/workflow-excerpt.ts', "test/review-entry-and-design-clarity-au.test.ts", "scripts/resolvers/preamble/generate-preamble-bash.ts", "scripts/resolvers/preamble/generate-completion-status.ts", 'scripts/resolvers/testing.ts', 'plan-eng-review/sections/**', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/review.ts', 'test/plan-review-cases.test.ts' @@ -1501,10 +1501,10 @@ export const LLM_JUDGE_TOUCHFILES: Record = { // /spec authored-spec quality (paid LLM-judge — periodic-tier). 'plan-design-review/SKILL.md passes': [ - "test/plan-scope-recovery-av.test.ts", + "test/fixtures/plan-scope-recovery-av.json", "test/fixtures/design-scope-checkpoint-at.json",'plan-design-review/SKILL.md', 'plan-design-review/SKILL.md.tmpl', 'test/skill-llm-eval.test.ts', 'test/helpers/workflow-judge-input.ts', 'test/helpers/workflow-judge-cache.ts', 'test/workflow-judge-cache.test.ts', 'scripts/eval-input-cache.ts', 'test/eval-input-cache.test.ts', 'test/workflow-judge-input.test.ts', 'test/helpers/workflow-excerpt.ts', - "test/design-scope-entry-aq.test.ts", + "test/plan-scope-selection.test.ts", "test/review-entry-and-design-clarity-au.test.ts", "scripts/resolvers/preamble/generate-preamble-bash.ts", "scripts/resolvers/preamble/generate-completion-status.ts", 'plan-design-review/sections/**', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/design.ts', 'scripts/resolvers/review.ts' diff --git a/test/model-overlay-fable-5.test.ts b/test/model-overlay-fable-5.test.ts deleted file mode 100644 index 537a4c03a..000000000 --- a/test/model-overlay-fable-5.test.ts +++ /dev/null @@ -1,55 +0,0 @@ -/** - * Fable 5 model overlay — gate-tier assertions on the family nudges. - * - * fable-5 inherits the claude base and adds Fable-family nudges: act when you - * have enough context (avoid over-planning), ground progress claims in tool - * results, assessment-vs-action boundaries, and delegate independent work. - */ -import { describe, test, expect } from 'bun:test'; -import * as fs from 'fs'; -import * as path from 'path'; -import type { TemplateContext } from '../scripts/resolvers/types'; -import { HOST_PATHS } from '../scripts/resolvers/types'; -import { generateModelOverlay } from '../scripts/resolvers/model-overlay'; - -function makeCtx(model: string): TemplateContext { - return { - skillName: 'test-skill', - tmplPath: 'test.tmpl', - host: 'claude', - paths: HOST_PATHS.claude, - preambleTier: 2, - model, - }; -} - -const ROOT = path.resolve(__dirname, '..'); - -describe('Fable 5 overlay — family nudges', () => { - test('raw fable-5.md contains the act-when-ready nudge', () => { - const raw = fs.readFileSync(path.join(ROOT, 'model-overlays/fable-5.md'), 'utf-8'); - expect(raw).toContain('Act when you have enough to act'); - }); - - test('resolved overlay inherits from claude base (INHERIT:claude)', () => { - const out = generateModelOverlay(makeCtx('fable-5')); - expect(out).toContain('Todo-list discipline'); - expect(out).toContain('subordinate'); - }); - - test('resolved overlay carries the Fable nudges', () => { - const out = generateModelOverlay(makeCtx('fable-5')); - expect(out).toContain('Act when you have enough to act'); - expect(out).toContain('Ground progress claims in evidence'); - }); - - test('resolved overlay has no unresolved INHERIT directive', () => { - const out = generateModelOverlay(makeCtx('fable-5')); - expect(out).not.toContain('{{INHERIT:'); - }); - - test('claude overlay (base) does not carry the Fable nudge', () => { - const out = generateModelOverlay(makeCtx('claude')); - expect(out).not.toContain('Act when you have enough to act'); - }); -}); diff --git a/test/model-overlay-gpt-5.6-sol.test.ts b/test/model-overlay-gpt-5.6-sol.test.ts deleted file mode 100644 index 6327f1a1f..000000000 --- a/test/model-overlay-gpt-5.6-sol.test.ts +++ /dev/null @@ -1,84 +0,0 @@ -import { describe, expect, test } from 'bun:test'; -import { resolveModel } from '../scripts/models'; -import { generateModelOverlay, readOverlay } from '../scripts/resolvers/model-overlay'; -import { generateCompletenessSection } from '../scripts/resolvers/preamble/generate-completeness-section'; -import { generateSetupCommand } from '../scripts/resolvers/utility'; -import type { TemplateContext } from '../scripts/resolvers/types'; - -function ctx(model: TemplateContext['model']): TemplateContext { - return { - skillName: 'investigate', - tmplPath: 'investigate/SKILL.md.tmpl', - host: 'codex', - paths: { - skillRoot: '$GSTACK_ROOT', - localSkillRoot: '.agents/skills/gstack', - binDir: '$GSTACK_BIN', - browseDir: '$GSTACK_BROWSE', - designDir: '$GSTACK_DESIGN', - makePdfDir: '$GSTACK_MAKE_PDF', - }, - preambleTier: 3, - model, - }; -} - -describe('GPT-5.6 Sol model profile', () => { - test('only the exact Sol ID selects the Sol profile', () => { - expect(resolveModel('gpt-5.6-sol')).toBe('gpt-5.6-sol'); - expect(resolveModel('gpt-5.6-terra')).toBe('gpt'); - expect(resolveModel('gpt-5.6-luna')).toBe('gpt'); - expect(resolveModel('gpt-5.6-sol-preview')).toBe('gpt'); - expect(resolveModel('gpt-5.7')).toBe('gpt'); - }); - - test('standalone overlay does not inherit generic GPT completion bias', () => { - const raw = readOverlay('gpt-5.6-sol'); - expect(raw).toContain('The explicit task is the lake'); - expect(raw).toContain('one clean relevant verification pass'); - expect(raw).toContain('report-only'); - expect(raw).not.toContain('{{INHERIT:gpt}}'); - expect(raw).not.toContain('make your best judgment and proceed'); - }); - - test('wrapper gives scope interpretation precedence but preserves concrete gates', () => { - const out = generateModelOverlay(ctx('gpt-5.6-sol')); - expect(out).toContain('disambiguate scope'); - expect(out).toContain('Concrete skill workflow steps'); - expect(out).toContain('Never use this patch to skip a concrete requirement'); - }); - - // The lake intro moved from a per-model render-time generator into - // bin/gstack-skill-start's one-time emission layer (token-reduction Phase 2). - // Sol's scope discipline is carried by the model overlay + completeness - // section (both still model-conditional and pinned here); the intro itself - // is a single display-once blurb emitted by the script. - test('completeness copy stays inside the explicit task boundary', () => { - const completeness = generateCompletenessSection(ctx('gpt-5.6-sol')); - expect(completeness).toContain("inside the user's explicit task boundary"); - expect(completeness).toContain('report them, do not implement them'); - expect(completeness).toContain('all relevant in-scope edge cases'); - }); - - test('generic GPT copy remains unchanged', () => { - const generic = generateModelOverlay(ctx('gpt')); - const completeness = generateCompletenessSection(ctx('gpt')); - expect(generic).toContain('make your best judgment and proceed'); - expect(completeness).toContain('the complete thing is the goal'); - }); - - test('terse mode still suppresses the completeness section for Sol', () => { - // Terse short-circuits before the Sol branch — a check-order flip would - // ship Sol completeness prose to terse users (a token regression). - expect(generateCompletenessSection({ ...ctx('gpt-5.6-sol'), explainLevel: 'terse' })).toBe(''); - }); -}); - -describe('SETUP_COMMAND resolver', () => { - test('claude keeps bare ./setup; every other host reinstalls itself', () => { - expect(generateSetupCommand({ ...ctx('claude'), host: 'claude' })).toBe('./setup'); - expect(generateSetupCommand({ ...ctx('gpt'), host: 'codex' })).toBe('./setup --host codex'); - expect(generateSetupCommand({ ...ctx('claude'), host: 'kiro' })).toBe('./setup --host kiro'); - expect(generateSetupCommand({ ...ctx('claude'), host: 'factory' })).toBe('./setup --host factory'); - }); -}); diff --git a/test/model-overlay-gpt-6-astra.test.ts b/test/model-overlay-gpt-6-astra.test.ts deleted file mode 100644 index 8810a6751..000000000 --- a/test/model-overlay-gpt-6-astra.test.ts +++ /dev/null @@ -1,41 +0,0 @@ -import { describe, expect, test } from 'bun:test'; -import * as fs from 'fs'; -import * as path from 'path'; -import { resolveModel } from '../scripts/models'; -import { generateModelOverlay } from '../scripts/resolvers/model-overlay'; -import type { TemplateContext } from '../scripts/resolvers/types'; - -function ctx(model: TemplateContext['model']): TemplateContext { - return { - skillName: 'investigate', - tmplPath: 'investigate/SKILL.md.tmpl', - host: 'codex', - paths: { - skillRoot: '$GSTACK_ROOT', - localSkillRoot: '.agents/skills/gstack', - binDir: '$GSTACK_BIN', - browseDir: '$GSTACK_BROWSE', - designDir: '$GSTACK_DESIGN', - makePdfDir: '$GSTACK_MAKE_PDF', - }, - preambleTier: 3, - model, - }; -} - -describe('GPT-6 Astra model profile', () => { - test('exact and suffixed Astra IDs select the Astra profile', () => { - expect(resolveModel('gpt-6-astra')).toBe('gpt-6-astra'); - expect(resolveModel('gpt-6-astra-2026-09-01')).toBe('gpt-6-astra'); - }); - - test('overlay inherits generic GPT guidance', () => { - const raw = fs.readFileSync(path.resolve(import.meta.dir, '..', 'model-overlays/gpt-6-astra.md'), 'utf-8'); - expect(raw).toContain('{{INHERIT:gpt}}'); - - const out = generateModelOverlay(ctx('gpt-6-astra')); - expect(out).toContain('make your best judgment and proceed'); - expect(out).toContain('Prefer decisive execution once scope is clear'); - expect(out).not.toContain('{{INHERIT:'); - }); -}); diff --git a/test/model-overlay-opus-4-7.test.ts b/test/model-overlay-opus-4-7.test.ts deleted file mode 100644 index 678ba0d6f..000000000 --- a/test/model-overlay-opus-4-7.test.ts +++ /dev/null @@ -1,97 +0,0 @@ -/** - * Opus 4.7 model overlay — gate-tier assertions on the pacing directive. - * - * v1.6.4.0 regressed plan-review cadence because the Opus 4.7 overlay - * carried a "Batch your questions" directive that physically rendered - * above the skill-level pacing rule. Opus 4.7 read top-to-bottom, - * absorbed batching as the ambient default, and stopped honoring the - * plan-review STOP directives. - * - * v1.7.0.0 replaces that block with "Pace questions to the skill" — - * one-question-at-a-time is now the default when the skill contains - * STOP directives; batching becomes the explicit exception. - * - * This test asserts: - * - The new "Pace questions" directive is present - * - The old "Batch your questions" directive is gone - * - The AUTO_DECIDE-compatible language survives (subordination, skill wins) - */ -import { describe, test, expect } from 'bun:test'; -import * as fs from 'fs'; -import * as path from 'path'; -import type { TemplateContext } from '../scripts/resolvers/types'; -import { HOST_PATHS } from '../scripts/resolvers/types'; -import { generateModelOverlay } from '../scripts/resolvers/model-overlay'; - -function makeCtx(model: string): TemplateContext { - return { - skillName: 'test-skill', - tmplPath: 'test.tmpl', - host: 'claude', - paths: HOST_PATHS.claude, - preambleTier: 2, - model, - }; -} - -const ROOT = path.resolve(__dirname, '..'); - -describe('Opus 4.7 overlay — pacing directive', () => { - test('raw opus-4-7.md contains "Pace questions to the skill"', () => { - const raw = fs.readFileSync( - path.join(ROOT, 'model-overlays/opus-4-7.md'), - 'utf-8', - ); - expect(raw).toContain('Pace questions to the skill'); - }); - - test('raw opus-4-7.md does NOT contain "Batch your questions" directive', () => { - const raw = fs.readFileSync( - path.join(ROOT, 'model-overlays/opus-4-7.md'), - 'utf-8', - ); - expect(raw).not.toContain('**Batch your questions.**'); - }); - - test('resolved overlay output contains "Pace questions to the skill"', () => { - const out = generateModelOverlay(makeCtx('opus-4-7')); - expect(out).toContain('Pace questions to the skill'); - }); - - test('resolved overlay inherits from claude base (INHERIT:claude)', () => { - const out = generateModelOverlay(makeCtx('opus-4-7')); - // The claude base contributes the subordination wrapper + Todo discipline - expect(out).toContain('Todo-list discipline'); - expect(out).toContain('subordinate'); - }); - - test('resolved overlay says skill STOP directives trigger one-per-turn pacing', () => { - const out = generateModelOverlay(makeCtx('opus-4-7')); - expect(out).toMatch(/STOP\. AskUserQuestion/); - expect(out).toMatch(/pace one question per turn|one question per turn/i); - }); - - test('resolved overlay requires AskUserQuestion as tool_use', () => { - const out = generateModelOverlay(makeCtx('opus-4-7')); - expect(out).toContain('tool_use'); - }); - - test('resolved overlay flags "obvious fix" findings still need user approval', () => { - const out = generateModelOverlay(makeCtx('opus-4-7')); - expect(out).toMatch(/obvious fix/i); - expect(out).toMatch(/user approval/i); - }); - - test('resolved overlay keeps Effort-match / Literal interpretation nudges', () => { - const out = generateModelOverlay(makeCtx('opus-4-7')); - expect(out).toContain('Effort-match the step'); - expect(out).toContain('Literal interpretation awareness'); - }); - - test('claude overlay (no INHERIT chain) does not carry the pacing directive', () => { - // Claude is the default overlay; opus-4-7 inherits FROM claude. - // The pacing directive belongs to opus-4-7 only. - const out = generateModelOverlay(makeCtx('claude')); - expect(out).not.toContain('Pace questions to the skill'); - }); -}); diff --git a/test/model-overlay-opus-4-8.test.ts b/test/model-overlay-opus-4-8.test.ts deleted file mode 100644 index e6fdda326..000000000 --- a/test/model-overlay-opus-4-8.test.ts +++ /dev/null @@ -1,92 +0,0 @@ -/** - * Opus 4.8 model overlay — gate-tier assertions on the pacing directive. - * - * opus-4-8 mirrors opus-4-7's Opus-4.x family nudges: it inherits the claude - * base and adds effort-matching, skill-paced questions (one-per-turn when the - * skill carries STOP directives), and complete-scope literal execution. - * - * This test asserts: - * - The "Pace questions to the skill" directive is present - * - The old "Batch your questions" directive is absent - * - The AUTO_DECIDE-compatible language survives (subordination, skill wins) - * - The claude base is inherited (INHERIT:claude) - */ -import { describe, test, expect } from 'bun:test'; -import * as fs from 'fs'; -import * as path from 'path'; -import type { TemplateContext } from '../scripts/resolvers/types'; -import { HOST_PATHS } from '../scripts/resolvers/types'; -import { generateModelOverlay } from '../scripts/resolvers/model-overlay'; - -function makeCtx(model: string): TemplateContext { - return { - skillName: 'test-skill', - tmplPath: 'test.tmpl', - host: 'claude', - paths: HOST_PATHS.claude, - preambleTier: 2, - model, - }; -} - -const ROOT = path.resolve(__dirname, '..'); - -describe('Opus 4.8 overlay — pacing directive', () => { - test('raw opus-4-8.md contains "Pace questions to the skill"', () => { - const raw = fs.readFileSync( - path.join(ROOT, 'model-overlays/opus-4-8.md'), - 'utf-8', - ); - expect(raw).toContain('Pace questions to the skill'); - }); - - test('raw opus-4-8.md does NOT contain "Batch your questions" directive', () => { - const raw = fs.readFileSync( - path.join(ROOT, 'model-overlays/opus-4-8.md'), - 'utf-8', - ); - expect(raw).not.toContain('**Batch your questions.**'); - }); - - test('resolved overlay output contains "Pace questions to the skill"', () => { - const out = generateModelOverlay(makeCtx('opus-4-8')); - expect(out).toContain('Pace questions to the skill'); - }); - - test('resolved overlay inherits from claude base (INHERIT:claude)', () => { - const out = generateModelOverlay(makeCtx('opus-4-8')); - // The claude base contributes the subordination wrapper + Todo discipline - expect(out).toContain('Todo-list discipline'); - expect(out).toContain('subordinate'); - }); - - test('resolved overlay says skill STOP directives trigger one-per-turn pacing', () => { - const out = generateModelOverlay(makeCtx('opus-4-8')); - expect(out).toMatch(/STOP\. AskUserQuestion/); - expect(out).toMatch(/pace one question per turn|one question per turn/i); - }); - - test('resolved overlay requires AskUserQuestion as tool_use', () => { - const out = generateModelOverlay(makeCtx('opus-4-8')); - expect(out).toContain('tool_use'); - }); - - test('resolved overlay flags "obvious fix" findings still need user approval', () => { - const out = generateModelOverlay(makeCtx('opus-4-8')); - expect(out).toMatch(/obvious fix/i); - expect(out).toMatch(/user approval/i); - }); - - test('resolved overlay keeps Effort-match / Literal interpretation nudges', () => { - const out = generateModelOverlay(makeCtx('opus-4-8')); - expect(out).toContain('Effort-match the step'); - expect(out).toContain('Literal interpretation awareness'); - }); - - test('claude overlay (no INHERIT chain) does not carry the pacing directive', () => { - // Claude is the default overlay; opus-4-8 inherits FROM claude. - // The pacing directive belongs to the opus-4-x overlays only. - const out = generateModelOverlay(makeCtx('claude')); - expect(out).not.toContain('Pace questions to the skill'); - }); -}); diff --git a/test/model-overlay-sonnet-5.test.ts b/test/model-overlay-sonnet-5.test.ts deleted file mode 100644 index 4665efc30..000000000 --- a/test/model-overlay-sonnet-5.test.ts +++ /dev/null @@ -1,56 +0,0 @@ -/** - * Sonnet 5 model overlay — gate-tier assertions on the family nudges. - * - * sonnet-5 inherits the claude base and adds Sonnet-5 family nudges: literal - * instruction following (state scope explicitly), scope work to the request - * (raise effort rather than prompting around shallow reasoning), and - * verbosity that tracks task complexity. - */ -import { describe, test, expect } from 'bun:test'; -import * as fs from 'fs'; -import * as path from 'path'; -import type { TemplateContext } from '../scripts/resolvers/types'; -import { HOST_PATHS } from '../scripts/resolvers/types'; -import { generateModelOverlay } from '../scripts/resolvers/model-overlay'; - -function makeCtx(model: string): TemplateContext { - return { - skillName: 'test-skill', - tmplPath: 'test.tmpl', - host: 'claude', - paths: HOST_PATHS.claude, - preambleTier: 2, - model, - }; -} - -const ROOT = path.resolve(__dirname, '..'); - -describe('Sonnet 5 overlay — family nudges', () => { - test('raw sonnet-5.md contains the literal-instructions nudge', () => { - const raw = fs.readFileSync(path.join(ROOT, 'model-overlays/sonnet-5.md'), 'utf-8'); - expect(raw).toContain('Instructions are read literally'); - }); - - test('resolved overlay inherits from claude base (INHERIT:claude)', () => { - const out = generateModelOverlay(makeCtx('sonnet-5')); - expect(out).toContain('Todo-list discipline'); - expect(out).toContain('subordinate'); - }); - - test('resolved overlay carries the Sonnet 5 nudges', () => { - const out = generateModelOverlay(makeCtx('sonnet-5')); - expect(out).toContain('Instructions are read literally'); - expect(out).toContain('Scope work to the request'); - }); - - test('resolved overlay has no unresolved INHERIT directive', () => { - const out = generateModelOverlay(makeCtx('sonnet-5')); - expect(out).not.toContain('{{INHERIT:'); - }); - - test('claude overlay (base) does not carry the Sonnet 5 nudge', () => { - const out = generateModelOverlay(makeCtx('claude')); - expect(out).not.toContain('Instructions are read literally'); - }); -}); diff --git a/test/model-overlays.test.ts b/test/model-overlays.test.ts new file mode 100644 index 000000000..e8ec21d19 --- /dev/null +++ b/test/model-overlays.test.ts @@ -0,0 +1,370 @@ +/** + * Model overlays: each family's resolved nudges, inheritance and model-ID routing. + */ +import { describe } from 'bun:test'; +import { test } from 'bun:test'; +import { expect } from 'bun:test'; +import * as fs from 'fs'; +import * as path from 'path'; +import type { TemplateContext } from '../scripts/resolvers/types'; +import { HOST_PATHS } from '../scripts/resolvers/types'; +import { generateModelOverlay } from '../scripts/resolvers/model-overlay'; +import { resolveModel } from '../scripts/models'; +import { readOverlay } from '../scripts/resolvers/model-overlay'; +import { generateCompletenessSection } from '../scripts/resolvers/preamble/generate-completeness-section'; +import { generateSetupCommand } from '../scripts/resolvers/utility'; + +describe('model-overlay-fable-5', () => { +function makeCtx(model: string): TemplateContext { + return { + skillName: 'test-skill', + tmplPath: 'test.tmpl', + host: 'claude', + paths: HOST_PATHS.claude, + preambleTier: 2, + model, + }; +} + +const ROOT = path.resolve(__dirname, '..'); + +describe('Fable 5 overlay — family nudges', () => { + test('raw fable-5.md contains the act-when-ready nudge', () => { + const raw = fs.readFileSync(path.join(ROOT, 'model-overlays/fable-5.md'), 'utf-8'); + expect(raw).toContain('Act when you have enough to act'); + }); + + test('resolved overlay inherits from claude base (INHERIT:claude)', () => { + const out = generateModelOverlay(makeCtx('fable-5')); + expect(out).toContain('Todo-list discipline'); + expect(out).toContain('subordinate'); + }); + + test('resolved overlay carries the Fable nudges', () => { + const out = generateModelOverlay(makeCtx('fable-5')); + expect(out).toContain('Act when you have enough to act'); + expect(out).toContain('Ground progress claims in evidence'); + }); + + test('resolved overlay has no unresolved INHERIT directive', () => { + const out = generateModelOverlay(makeCtx('fable-5')); + expect(out).not.toContain('{{INHERIT:'); + }); + + test('claude overlay (base) does not carry the Fable nudge', () => { + const out = generateModelOverlay(makeCtx('claude')); + expect(out).not.toContain('Act when you have enough to act'); + }); +}); +}); + +describe('model-overlay-gpt-5.6-sol', () => { +function ctx(model: TemplateContext['model']): TemplateContext { + return { + skillName: 'investigate', + tmplPath: 'investigate/SKILL.md.tmpl', + host: 'codex', + paths: { + skillRoot: '$GSTACK_ROOT', + localSkillRoot: '.agents/skills/gstack', + binDir: '$GSTACK_BIN', + browseDir: '$GSTACK_BROWSE', + designDir: '$GSTACK_DESIGN', + makePdfDir: '$GSTACK_MAKE_PDF', + }, + preambleTier: 3, + model, + }; +} + +describe('GPT-5.6 Sol model profile', () => { + test('only the exact Sol ID selects the Sol profile', () => { + expect(resolveModel('gpt-5.6-sol')).toBe('gpt-5.6-sol'); + expect(resolveModel('gpt-5.6-terra')).toBe('gpt'); + expect(resolveModel('gpt-5.6-luna')).toBe('gpt'); + expect(resolveModel('gpt-5.6-sol-preview')).toBe('gpt'); + expect(resolveModel('gpt-5.7')).toBe('gpt'); + }); + + test('standalone overlay does not inherit generic GPT completion bias', () => { + const raw = readOverlay('gpt-5.6-sol'); + expect(raw).toContain('The explicit task is the lake'); + expect(raw).toContain('one clean relevant verification pass'); + expect(raw).toContain('report-only'); + expect(raw).not.toContain('{{INHERIT:gpt}}'); + expect(raw).not.toContain('make your best judgment and proceed'); + }); + + test('wrapper gives scope interpretation precedence but preserves concrete gates', () => { + const out = generateModelOverlay(ctx('gpt-5.6-sol')); + expect(out).toContain('disambiguate scope'); + expect(out).toContain('Concrete skill workflow steps'); + expect(out).toContain('Never use this patch to skip a concrete requirement'); + }); + + // The lake intro moved from a per-model render-time generator into + // bin/gstack-skill-start's one-time emission layer (token-reduction Phase 2). + // Sol's scope discipline is carried by the model overlay + completeness + // section (both still model-conditional and pinned here); the intro itself + // is a single display-once blurb emitted by the script. + test('completeness copy stays inside the explicit task boundary', () => { + const completeness = generateCompletenessSection(ctx('gpt-5.6-sol')); + expect(completeness).toContain("inside the user's explicit task boundary"); + expect(completeness).toContain('report them, do not implement them'); + expect(completeness).toContain('all relevant in-scope edge cases'); + }); + + test('generic GPT copy remains unchanged', () => { + const generic = generateModelOverlay(ctx('gpt')); + const completeness = generateCompletenessSection(ctx('gpt')); + expect(generic).toContain('make your best judgment and proceed'); + expect(completeness).toContain('the complete thing is the goal'); + }); + + test('terse mode still suppresses the completeness section for Sol', () => { + // Terse short-circuits before the Sol branch — a check-order flip would + // ship Sol completeness prose to terse users (a token regression). + expect(generateCompletenessSection({ ...ctx('gpt-5.6-sol'), explainLevel: 'terse' })).toBe(''); + }); +}); + +describe('SETUP_COMMAND resolver', () => { + test('claude keeps bare ./setup; every other host reinstalls itself', () => { + expect(generateSetupCommand({ ...ctx('claude'), host: 'claude' })).toBe('./setup'); + expect(generateSetupCommand({ ...ctx('gpt'), host: 'codex' })).toBe('./setup --host codex'); + expect(generateSetupCommand({ ...ctx('claude'), host: 'kiro' })).toBe('./setup --host kiro'); + expect(generateSetupCommand({ ...ctx('claude'), host: 'factory' })).toBe('./setup --host factory'); + }); +}); +}); + +describe('model-overlay-gpt-6-astra', () => { +function ctx(model: TemplateContext['model']): TemplateContext { + return { + skillName: 'investigate', + tmplPath: 'investigate/SKILL.md.tmpl', + host: 'codex', + paths: { + skillRoot: '$GSTACK_ROOT', + localSkillRoot: '.agents/skills/gstack', + binDir: '$GSTACK_BIN', + browseDir: '$GSTACK_BROWSE', + designDir: '$GSTACK_DESIGN', + makePdfDir: '$GSTACK_MAKE_PDF', + }, + preambleTier: 3, + model, + }; +} + +describe('GPT-6 Astra model profile', () => { + test('exact and suffixed Astra IDs select the Astra profile', () => { + expect(resolveModel('gpt-6-astra')).toBe('gpt-6-astra'); + expect(resolveModel('gpt-6-astra-2026-09-01')).toBe('gpt-6-astra'); + }); + + test('overlay inherits generic GPT guidance', () => { + const raw = fs.readFileSync(path.resolve(import.meta.dir, '..', 'model-overlays/gpt-6-astra.md'), 'utf-8'); + expect(raw).toContain('{{INHERIT:gpt}}'); + + const out = generateModelOverlay(ctx('gpt-6-astra')); + expect(out).toContain('make your best judgment and proceed'); + expect(out).toContain('Prefer decisive execution once scope is clear'); + expect(out).not.toContain('{{INHERIT:'); + }); +}); +}); + +describe('model-overlay-opus-4-7', () => { +function makeCtx(model: string): TemplateContext { + return { + skillName: 'test-skill', + tmplPath: 'test.tmpl', + host: 'claude', + paths: HOST_PATHS.claude, + preambleTier: 2, + model, + }; +} + +const ROOT = path.resolve(__dirname, '..'); + +describe('Opus 4.7 overlay — pacing directive', () => { + test('raw opus-4-7.md contains "Pace questions to the skill"', () => { + const raw = fs.readFileSync( + path.join(ROOT, 'model-overlays/opus-4-7.md'), + 'utf-8', + ); + expect(raw).toContain('Pace questions to the skill'); + }); + + test('raw opus-4-7.md does NOT contain "Batch your questions" directive', () => { + const raw = fs.readFileSync( + path.join(ROOT, 'model-overlays/opus-4-7.md'), + 'utf-8', + ); + expect(raw).not.toContain('**Batch your questions.**'); + }); + + test('resolved overlay output contains "Pace questions to the skill"', () => { + const out = generateModelOverlay(makeCtx('opus-4-7')); + expect(out).toContain('Pace questions to the skill'); + }); + + test('resolved overlay inherits from claude base (INHERIT:claude)', () => { + const out = generateModelOverlay(makeCtx('opus-4-7')); + // The claude base contributes the subordination wrapper + Todo discipline + expect(out).toContain('Todo-list discipline'); + expect(out).toContain('subordinate'); + }); + + test('resolved overlay says skill STOP directives trigger one-per-turn pacing', () => { + const out = generateModelOverlay(makeCtx('opus-4-7')); + expect(out).toMatch(/STOP\. AskUserQuestion/); + expect(out).toMatch(/pace one question per turn|one question per turn/i); + }); + + test('resolved overlay requires AskUserQuestion as tool_use', () => { + const out = generateModelOverlay(makeCtx('opus-4-7')); + expect(out).toContain('tool_use'); + }); + + test('resolved overlay flags "obvious fix" findings still need user approval', () => { + const out = generateModelOverlay(makeCtx('opus-4-7')); + expect(out).toMatch(/obvious fix/i); + expect(out).toMatch(/user approval/i); + }); + + test('resolved overlay keeps Effort-match / Literal interpretation nudges', () => { + const out = generateModelOverlay(makeCtx('opus-4-7')); + expect(out).toContain('Effort-match the step'); + expect(out).toContain('Literal interpretation awareness'); + }); + + test('claude overlay (no INHERIT chain) does not carry the pacing directive', () => { + // Claude is the default overlay; opus-4-7 inherits FROM claude. + // The pacing directive belongs to opus-4-7 only. + const out = generateModelOverlay(makeCtx('claude')); + expect(out).not.toContain('Pace questions to the skill'); + }); +}); +}); + +describe('model-overlay-opus-4-8', () => { +function makeCtx(model: string): TemplateContext { + return { + skillName: 'test-skill', + tmplPath: 'test.tmpl', + host: 'claude', + paths: HOST_PATHS.claude, + preambleTier: 2, + model, + }; +} + +const ROOT = path.resolve(__dirname, '..'); + +describe('Opus 4.8 overlay — pacing directive', () => { + test('raw opus-4-8.md contains "Pace questions to the skill"', () => { + const raw = fs.readFileSync( + path.join(ROOT, 'model-overlays/opus-4-8.md'), + 'utf-8', + ); + expect(raw).toContain('Pace questions to the skill'); + }); + + test('raw opus-4-8.md does NOT contain "Batch your questions" directive', () => { + const raw = fs.readFileSync( + path.join(ROOT, 'model-overlays/opus-4-8.md'), + 'utf-8', + ); + expect(raw).not.toContain('**Batch your questions.**'); + }); + + test('resolved overlay output contains "Pace questions to the skill"', () => { + const out = generateModelOverlay(makeCtx('opus-4-8')); + expect(out).toContain('Pace questions to the skill'); + }); + + test('resolved overlay inherits from claude base (INHERIT:claude)', () => { + const out = generateModelOverlay(makeCtx('opus-4-8')); + // The claude base contributes the subordination wrapper + Todo discipline + expect(out).toContain('Todo-list discipline'); + expect(out).toContain('subordinate'); + }); + + test('resolved overlay says skill STOP directives trigger one-per-turn pacing', () => { + const out = generateModelOverlay(makeCtx('opus-4-8')); + expect(out).toMatch(/STOP\. AskUserQuestion/); + expect(out).toMatch(/pace one question per turn|one question per turn/i); + }); + + test('resolved overlay requires AskUserQuestion as tool_use', () => { + const out = generateModelOverlay(makeCtx('opus-4-8')); + expect(out).toContain('tool_use'); + }); + + test('resolved overlay flags "obvious fix" findings still need user approval', () => { + const out = generateModelOverlay(makeCtx('opus-4-8')); + expect(out).toMatch(/obvious fix/i); + expect(out).toMatch(/user approval/i); + }); + + test('resolved overlay keeps Effort-match / Literal interpretation nudges', () => { + const out = generateModelOverlay(makeCtx('opus-4-8')); + expect(out).toContain('Effort-match the step'); + expect(out).toContain('Literal interpretation awareness'); + }); + + test('claude overlay (no INHERIT chain) does not carry the pacing directive', () => { + // Claude is the default overlay; opus-4-8 inherits FROM claude. + // The pacing directive belongs to the opus-4-x overlays only. + const out = generateModelOverlay(makeCtx('claude')); + expect(out).not.toContain('Pace questions to the skill'); + }); +}); +}); + +describe('model-overlay-sonnet-5', () => { +function makeCtx(model: string): TemplateContext { + return { + skillName: 'test-skill', + tmplPath: 'test.tmpl', + host: 'claude', + paths: HOST_PATHS.claude, + preambleTier: 2, + model, + }; +} + +const ROOT = path.resolve(__dirname, '..'); + +describe('Sonnet 5 overlay — family nudges', () => { + test('raw sonnet-5.md contains the literal-instructions nudge', () => { + const raw = fs.readFileSync(path.join(ROOT, 'model-overlays/sonnet-5.md'), 'utf-8'); + expect(raw).toContain('Instructions are read literally'); + }); + + test('resolved overlay inherits from claude base (INHERIT:claude)', () => { + const out = generateModelOverlay(makeCtx('sonnet-5')); + expect(out).toContain('Todo-list discipline'); + expect(out).toContain('subordinate'); + }); + + test('resolved overlay carries the Sonnet 5 nudges', () => { + const out = generateModelOverlay(makeCtx('sonnet-5')); + expect(out).toContain('Instructions are read literally'); + expect(out).toContain('Scope work to the request'); + }); + + test('resolved overlay has no unresolved INHERIT directive', () => { + const out = generateModelOverlay(makeCtx('sonnet-5')); + expect(out).not.toContain('{{INHERIT:'); + }); + + test('claude overlay (base) does not carry the Sonnet 5 nudge', () => { + const out = generateModelOverlay(makeCtx('claude')); + expect(out).not.toContain('Instructions are read literally'); + }); +}); +}); diff --git a/test/native-auto-decide.test.ts b/test/native-auto-decide.test.ts index a99e1df2d..a8753a905 100644 --- a/test/native-auto-decide.test.ts +++ b/test/native-auto-decide.test.ts @@ -6,6 +6,19 @@ import { findNativeAutoDecision } from './helpers/native-auto-decide'; import { classifyVisible } from './helpers/claude-pty-runner'; import { readPlanCountTranscript, type NativePublicToolEvent } from './helpers/plan-count-transcript'; import { E2E_TOUCHFILES, selectTests } from './helpers/touchfiles'; +import capture_auto_decide_current_declaration from './fixtures/auto-decide-current-declaration-6aef.json'; +import capture_auto_decide_explanatory_mode from './fixtures/auto-decide-explanatory-mode-043a.json'; +import captured749_auto_decide_explanatory_mode from './fixtures/auto-decide-explanatory-mode-749df.json'; +import annotations_auto_decide_explanatory_mode from './fixtures/native-auto-decide-ag.json'; +import capture_auto_decide_recommendation_scope from './fixtures/auto-decide-recommendation-361c.json'; +import captured_auto_decide_saved_ai from './fixtures/auto-decide-saved-ai.json'; +import retry_auto_decide_saved_ai from './fixtures/auto-decide-retry-ai.json'; +import capture_auto_decide_structured from './fixtures/auto-decide-structured-77.json'; +import completedModeCapture_auto_decide_structured from './fixtures/auto-decide-completed-mode-f359.json'; +import capture_auto_decide_target_identity from './fixtures/auto-decide-target-361c.json'; +import { bindAutoDecisionState } from './helpers/auto-decision-state'; +import capture_auto_decision_state from './fixtures/auto-decide-state-cab3.json'; +import { describe } from 'bun:test'; const captured = JSON.parse(fs.readFileSync(path.join(import.meta.dir,'fixtures/native-auto-decide-ag.json'),'utf8')); const clone = (i=0) => structuredClone(captured.attempts[i]); const verdict = (f=clone()) => findNativeAutoDecision(f.transcript,f.tools,f.options); @@ -104,3 +117,1068 @@ test('new native annotation dependencies retain all existing observation caller expect(selectTests([file],E2E_TOUCHFILES).selected).toContain('auto-decide-preserved'); } }); + +describe('auto-decide-current-declaration', () => { +const capture = capture_auto_decide_current_declaration; +const clone = () => structuredClone(capture.retry) as any; +const decide = (f: any) => findNativeAutoDecision(f.transcript, f.tools, f.options); +const message = (f: any) => f.transcript.assistantMessages.find((m: any) => + m.timestamp === '2026-09-16T23:23:28.931Z'); +const use = (f: any) => f.tools.find((e: any) => e.kind === 'use' && + e.input?.command?.includes('gstack-skill-start')); +const ack = (f: any) => f.tools.find((e: any) => e.kind === 'result' && e.toolUseId === use(f).toolUseId); + +test('actual current Decision declaration completes the retained owned retry', () => { + const f = clone(); + const result = decide(f); + expect(result?.option).toBe('HOLD SCOPE'); + expect(result?.annotation).toBe(message(f).text); + expect(result?.stateRecord).toEqual(f.options.stateEvidence.records[0]); + expect(result?.preambleToolUseId).toBe(use(f).toolUseId); + expect(result?.skillToolUseId).toBeUndefined(); + expect(result?.questionLogToolUseId).toBeUndefined(); +}); + +test('literal first declaration form is supported by the authenticated retry context', () => { + const f = clone(); + // This checks representation only. The first attempt's state was deleted; + // transplanting its text grants no first-attempt ownership or verdict credit. + message(f).text = capture.firstDeclaration.text; + expect(decide(f)?.option).toBe('HOLD SCOPE'); + delete f.options.stateEvidence; + f.tools = []; + expect(decide(f)).toBeNull(); +}); + +const modes = ['HOLD SCOPE', 'SCOPE EXPANSION', 'SELECTIVE EXPANSION', 'SCOPE REDUCTION']; +for (const mode of modes) { + for (const label of ['Decision', 'Decision: review mode is', 'Mode']) { + for (const target of ['for this draft.', 'for the current review (saved preference).']) { + const text = `${label}${label.includes(':') ? '' : ':'} ${mode} ${target}`; + test(`one current full mode owns its target clause: ${text}`, () => { + const f = clone(); + Object.assign(f.options.stateEvidence.records[0], { user_choice: mode, recommended: mode }); + message(f).text = text; + expect(decide(f)?.option).toBe(mode); + }); + } + } +} + +const invalidDeclarations = [ + 'Decision: HOLD SCOPELESS for this draft.', + 'Decision: HOLD for this draft.', + 'Decision: SCOPE for this draft.', + 'Decision: SELECTIVE for this draft.', + 'Decision: review mode is CUSTOM MODE for this draft.', + 'Decision: HOLD SCOPE?', + 'Decision: HOLD SCOPE for', + 'Decision: HOLD SCOPE for (', + 'Decision: HOLD SCOPE for this draft (unfinished.', + 'Decision: HOLD SCOPE for this draft (unbalanced)).', + 'Decision: HOLD SCOPE for this draft or SCOPE EXPANSION.', + 'Decision: HOLD SCOPE for this draft. Instead choose SCOPE REDUCTION.', + 'Decision: HOLD SCOPE for this draft; SELECTIVE_EXPANSION.', + 'Decision: HOLD SCOPE for this draft, if approved.', + 'Decision: HOLD SCOPE for this draft, pending approval.', + 'Decision: HOLD SCOPE for this draft, not yet selected.', + 'Decision: HOLD SCOPE for this draft, withdrawn.', + 'Decision: HOLD SCOPE for this draft; the selected mode is not HOLD SCOPE.', + 'Decision: HOLD SCOPE for plan-eng-review.', + 'Decision: HOLD SCOPE for another draft.', + 'Decision: HOLD SCOPE for a future review.', + 'Decision: HOLD SCOPE for this future review.', + 'Decision pending: HOLD SCOPE for this draft.', + 'Decision: review mode is not selected.', + 'Decision: not HOLD SCOPE for this draft.', +]; +for (const text of invalidDeclarations) { + test(`unsupported current declaration cannot complete a decision: ${text}`, () => { + const f = clone(); message(f).text = text; + expect(decide(f)).toBeNull(); + }); + test(`unsupported current declaration retracts the earlier decision: ${text}`, () => { + const f = clone(); message(f).text += `\n\nCorrection: ${text}`; + expect(decide(f)).toBeNull(); + }); +} + +for (const prefix of ['> ', ' ', '"', '`']) { + test(`quoted current-mode syntax does not declare or retract: ${JSON.stringify(prefix)}`, () => { + const f = clone(); + const quote = (value: string) => prefix + value + (['"', '`'].includes(prefix) ? prefix : ''); + message(f).text = quote('Decision: HOLD SCOPE for this draft.'); + expect(decide(f)).toBeNull(); + message(f).text = clone().transcript.assistantMessages.at(-1).text + '\n\n' + + quote('Decision: SCOPE EXPANSION for this draft.'); + expect(decide(f)?.option).toBe('HOLD SCOPE'); + }); +} +for (const text of [ + 'Example:\nDecision: HOLD SCOPE for this draft.', + 'Historical transcript:\nDecision: HOLD SCOPE for this draft.', + 'Previous decision:\nDecision: HOLD SCOPE for this draft.', + '```text\nDecision: HOLD SCOPE for this draft.\n```', + 'If approved, Decision: HOLD SCOPE for this draft.', + 'Not a decision: HOLD SCOPE for this draft.', +]) test(`unasserted declaration provides no mode: ${JSON.stringify(text)}`, () => { + const f = clone(); message(f).text = text; + expect(decide(f)).toBeNull(); +}); + +for (const label of ['Decision', 'Decision: review mode is', 'Mode']) { + test(`later matching current declaration retains the owned mode: ${label}`, () => { + const f = clone(); message(f).text += `\n\nUpdate: ${label}${label.includes(':') ? '' : ':'} HOLD SCOPE for this draft.`; + expect(decide(f)?.option).toBe('HOLD SCOPE'); + }); + test(`later conflicting current declaration retracts the owned mode: ${label}`, () => { + const f = clone(); message(f).text += `\n\nUpdate: ${label}${label.includes(':') ? '' : ':'} SCOPE EXPANSION for this draft.`; + expect(decide(f)).toBeNull(); + }); +} + +test('a separately scoped non-mode decision does not retract the review mode', () => { + const f = clone(); message(f).text += '\n\nDecision: publish the audit log.'; + expect(decide(f)?.option).toBe('HOLD SCOPE'); +}); + +test('the completed native preference and log ACK retain their independent authority', () => { + const f = clone(); delete f.options.stateEvidence; + const result = decide(f); + expect(result?.option).toBe('HOLD SCOPE'); + expect(result?.stateRecord).toBeUndefined(); + expect(result?.preferenceToolUseId).toBeDefined(); + expect(result?.questionLogToolUseId).toBeDefined(); +}); + +test('target-clause capitalization and a negative non-mode explanation remain valid', () => { + const f = clone(); message(f).text = 'Decision: HOLD SCOPE For this draft, not for implementation.'; + expect(decide(f)?.option).toBe('HOLD SCOPE'); +}); + +test('a named target must match the complete audit target, including dotted identifiers', () => { + const f = clone(); + f.options.stateEvidence.records[0].question_summary = 'Select review mode for PLAN.md'; + message(f).text = 'Decision: HOLD SCOPE for PLAN.md.'; + expect(decide(f)?.option).toBe('HOLD SCOPE'); + message(f).text = 'Decision: HOLD SCOPE for PLAN.other.'; + expect(decide(f)).toBeNull(); +}); + +for (const target of ['a future review', 'the previous review', 'another draft', 'the next invocation', 'an example plan']) { + test(`even a matching audit cannot make an explicitly noncurrent target current: ${target}`, () => { + const f = clone(); + f.options.stateEvidence.records[0].question_summary = `Select review mode for ${target}`; + message(f).text = `Decision: HOLD SCOPE for ${target}.`; + expect(decide(f)).toBeNull(); + }); +} +for (const target of ['future.md', 'previous-review.md', 'another.plan.md']) { + test(`owned literal filename remains a current target: ${target}`, () => { + const f = clone(); + f.options.stateEvidence.records[0].question_summary = `Select review mode for ${target}`; + message(f).text = `Decision: HOLD SCOPE for ${target}.`; + expect(decide(f)?.option).toBe('HOLD SCOPE'); + }); +} + +for (const [name, mutate] of Object.entries({ + 'missing state and native log ACK': (f: any) => { + delete f.options.stateEvidence; + const log = f.tools.find((e: any) => e.kind === 'use' && e.input?.command?.includes('gstack-question-log')); + f.tools = f.tools.filter((e: any) => e.kind !== 'result' || e.toolUseId !== log.toolUseId); + }, + 'empty owned log': (f: any) => { f.options.stateEvidence.records = []; }, + 'duplicate owned log': (f: any) => { f.options.stateEvidence.records.push({ ...f.options.stateEvidence.records[0] }); }, + 'wrong preference and native check': (f: any) => { + f.options.stateEvidence.preference = 'always-ask'; + const check = f.tools.find((e: any) => e.kind === 'use' && e.input?.command?.includes('gstack-question-preference')); + f.tools.find((e: any) => e.kind === 'result' && e.toolUseId === check.toolUseId).content = 'ASK\nEXIT: 0'; + }, + 'foreign log session': (f: any) => { f.options.stateEvidence.records[0].session_id = 'foreign'; }, + 'foreign log skill': (f: any) => { f.options.stateEvidence.records[0].skill = 'plan-eng-review'; }, + 'different logged choice': (f: any) => { f.options.stateEvidence.records[0].user_choice = 'SCOPE EXPANSION'; }, + 'different recommendation': (f: any) => { f.options.stateEvidence.records[0].recommended = 'SCOPE EXPANSION'; }, + 'nonautomatic log': (f: any) => { f.options.stateEvidence.records[0].auto_decided = false; }, + 'foreign native session': (f: any) => { f.options.sessionId = 'foreign'; }, + 'failed preamble': (f: any) => { ack(f).isError = true; }, + 'missing preamble ACK': (f: any) => { f.tools = f.tools.filter((e: any) => e !== ack(f)); }, + 'duplicate preamble ACK': (f: any) => { f.tools.push({ ...ack(f) }); }, + 'disabled tuning': (f: any) => { ack(f).content = ack(f).content.replace('QUESTION_TUNING: true', 'QUESTION_TUNING: false'); }, + 'old log': (f: any) => { f.options.stateEvidence.records[0].ts = new Date(f.options.commandStartedAt - 1).toISOString(); }, + 'future log': (f: any) => { f.options.stateEvidence.records[0].ts = new Date(f.options.now + 1).toISOString(); }, + 'public decision before log': (f: any) => { message(f).timestamp = new Date(Date.parse(f.options.stateEvidence.records[0].ts) - 1).toISOString(); }, + 'native question': (f: any) => { f.transcript.calls.push({ sessionId: f.options.sessionId }); }, + 'prose question': (f: any) => { f.options.proseQuestionObserved = true; }, + 'current withdrawal': (f: any) => { message(f).text += '\n\nI withdraw this decision.'; }, +})) test(`current Decision syntax retains ${name} rejection`, () => { + const f = clone(); mutate(f); expect(decide(f)).toBeNull(); +}); +}); + +describe('auto-decide-explanatory-mode', () => { +const capture = capture_auto_decide_explanatory_mode; +const captured749 = captured749_auto_decide_explanatory_mode; +const annotations = annotations_auto_decide_explanatory_mode; +const clone = () => structuredClone(capture) as any; +const declaration = (f: any) => f.transcript.assistantMessages.find((m: any) => + m.text.startsWith('**Review mode: HOLD SCOPE** —')); +const decide = (f: any) => findNativeAutoDecision(f.transcript, f.tools, f.options); + +function witnessed() { + const f = clone(); + const use = f.tools.find((e: any) => e.input?.command?.includes('gstack-question-log')); + // Synthetic owned-file witness from the exact literal request. The original + // file was not retained; its failed paid attempt remains failed. + const record = JSON.parse(/gstack-question-log '(\{[^\n]*\})'/.exec(use.input.command)![1]!); + record.source = 'agent'; + record.ts = f.tools.find((e: any) => e.kind === 'result' && e.toolUseId === use.toolUseId).timestamp; + f.options.stateEvidence = { questionId: 'plan-ceo-review-mode', preference: 'never-ask', records: [record] }; + return f; +} + +test('original public events alone cannot authenticate the unretained owned append', () => { + expect(decide(clone())).toBeNull(); +}); + +test('exact completed announcement agrees with an authenticated owned append', () => { + const f = witnessed(); + const result = decide(f); + expect(result?.option).toBe('HOLD SCOPE'); + expect(result?.annotation).toBe(declaration(f).text); + expect(result?.stateRecord).toEqual(f.options.stateEvidence.records[0]); + expect(result?.questionLogToolUseId).toBeUndefined(); +}); + +const separators = ['. ', ', ', '; ', ': ', ' — ', ' – ', ' - ']; +for (const separator of separators) { + test(`complete mode with separated explanation ${JSON.stringify(separator)}`, () => { + const f = witnessed(); + declaration(f).text = `Review mode: HOLD SCOPE${separator}selected from the saved preference.`; + expect(decide(f)?.option).toBe('HOLD SCOPE'); + }); + test(`same later completed mode retains its explanation ${JSON.stringify(separator)}`, () => { + const f = witnessed(); + declaration(f).text += `\n\nMode decision completed: HOLD SCOPE${separator}selected from the saved preference.`; + expect(decide(f)?.option).toBe('HOLD SCOPE'); + }); + test(`different later completed mode withdraws the choice ${JSON.stringify(separator)}`, () => { + const f = witnessed(); + declaration(f).text += `\n\nCorrection: Mode: SCOPE EXPANSION${separator}selected from the saved preference.`; + expect(decide(f)).toBeNull(); + }); + test(`conditional explanation never completes the mode ${JSON.stringify(separator)}`, () => { + const f = witnessed(); + declaration(f).text = `Review mode: HOLD SCOPE${separator}if approved.`; + expect(decide(f)).toBeNull(); + }); + test(`later conditional explanation withdraws the choice ${JSON.stringify(separator)}`, () => { + const f = witnessed(); + declaration(f).text += `\n\nMode: HOLD SCOPE${separator}pending approval.`; + expect(decide(f)).toBeNull(); + }); +} + +for (const value of ['HOLD SCOPELESS', 'HOLD SCOPE SCOPE EXPANSION', 'HOLD SCOPE / SCOPE EXPANSION', + 'HOLD SCOPE?', 'HOLD SCOPE selected from my preference', 'HOLD SCOPE—if approved', 'HOLD SCOPE - ']) { + test(`incomplete or ambiguous mode is not a declaration: ${value}`, () => { + const f = witnessed(); declaration(f).text = `Mode: ${value}`; + expect(decide(f)).toBeNull(); + }); + test(`incomplete current field retracts a previous mode: ${value}`, () => { + const f = witnessed(); declaration(f).text += `\n\nMode: ${value}`; + expect(decide(f)).toBeNull(); + }); +} + +for (const [name, mutate] of Object.entries({ + 'missing append': (f: any) => { f.options.stateEvidence.records = []; }, + 'foreign session': (f: any) => { f.options.stateEvidence.records[0].session_id = 'foreign'; }, + 'different logged mode': (f: any) => { f.options.stateEvidence.records[0].user_choice = 'SCOPE EXPANSION'; }, + 'native question': (f: any) => { f.transcript.calls.push({ sessionId: f.options.sessionId }); }, + 'unfinished declaration': (f: any) => { declaration(f).text = 'Mode decision pending: HOLD SCOPE — saved preference.'; }, + 'withdrawn decision': (f: any) => { declaration(f).text += '\n\nI withdraw this decision.'; }, + 'quoted declaration': (f: any) => { declaration(f).text = '> Mode: HOLD SCOPE — saved preference.'; }, + 'example declaration': (f: any) => { declaration(f).text = 'Example:\nMode: HOLD SCOPE — saved preference.'; }, +})) test(`explanatory announcement still rejects ${name}`, () => { + const f = witnessed(); mutate(f); expect(decide(f)).toBeNull(); +}); + +test('quoted historical correction does not withdraw the current completed mode', () => { + const f = witnessed(); + declaration(f).text += '\n\n> Mode: SCOPE EXPANSION — a historical example.'; + expect(decide(f)?.option).toBe('HOLD SCOPE'); +}); + +test('generic Skill annotations retain their existing non-CEO mode vocabulary', () => { + const f = structuredClone(annotations.attempts[0]) as any; + f.options.skillName = 'office-hours'; + f.tools.find((e: any) => e.kind === 'use' && e.name === 'Skill').input.skill = 'office-hours'; + const message = f.transcript.assistantMessages.find((m: any) => m.text.startsWith('Auto-decided')); + message.text = 'Auto-decided workflow → **Builder** (your preference). Change with /plan-tune.\n\nMode: Builder (saved preference).'; + expect(decide(f)?.option).toBe('Builder'); + message.text += '\n\nMode: Startup (saved preference).'; + expect(decide(f)).toBeNull(); +}); + +test('retained retry messages alone cannot authenticate missing tool and file evidence', () => { + const retry = capture.retryObservation; + expect(findNativeAutoDecision(retry.transcript, [], retry.options)).toBeNull(); +}); + +test('exact retry prose accepts the optional decision label in an owned context', () => { + const f = witnessed(); + // Only the text is replayed. Session/time and owned witness belong to the + // first fixture; this is not a reconstruction or promotion of the retry. + declaration(f).text = capture.retryObservation.transcript.assistantMessages.find(m => + m.text.startsWith('**Mode decision:'))!.text; + expect(decide(f)?.option).toBe('HOLD SCOPE'); +}); + +for (const field of ['Mode', 'Mode decision', 'Review mode', 'Review mode decision']) { + test(`a completed field does not require a separate status word: ${field}`, () => { + const f = witnessed(); declaration(f).text = `${field}: HOLD SCOPE (saved preference).`; + expect(decide(f)?.option).toBe('HOLD SCOPE'); + }); + test(`a later matching field does not withdraw its choice: ${field}`, () => { + const f = witnessed(); declaration(f).text += `\n\n${field}: HOLD SCOPE (saved preference).`; + expect(decide(f)?.option).toBe('HOLD SCOPE'); + }); + test(`a conflicting later field still withdraws its choice: ${field}`, () => { + const f = witnessed(); declaration(f).text += `\n\n${field}: SCOPE EXPANSION (saved preference).`; + expect(decide(f)).toBeNull(); + }); +} + +const clone749 = () => structuredClone(captured749) as any; +const declaration749 = (f: any) => f.transcript.assistantMessages.find((m: any) => + m.timestamp === '2026-09-16T12:13:02.513Z'); + +test('actual 749 public declaration agrees with its retained owned append', () => { + // Exact public tools, final declaration and owned log were captured while the + // paid observer was still waiting. This free replay does not promote that run. + const f = clone749(); + const result = decide(f); + expect(result?.option).toBe('HOLD SCOPE'); + expect(result?.annotation).toBe(declaration749(f).text); + expect(result?.stateRecord).toEqual(f.options.stateEvidence.records[0]); +}); + +const explanatoryTails = [ + ' (saved preference; no prompt required). The review remains paused.', + ' (saved preference (confirmed for this project); no prompt required). The review remains paused.', + ' (saved preference), recorded for this invocation.', + ' (saved preference): recorded for this invocation.', + ' (saved preference) — recorded for this invocation.', + ' (saved preference)\nThe review remains paused.', + '. Selected from the saved preference (recorded).', + '; selected from the saved preference (recorded).', +]; +for (const tail of explanatoryTails) { + test(`balanced explanation with following prose is a complete declaration: ${JSON.stringify(tail)}`, () => { + const f = clone749(); declaration749(f).text = `Mode decision: HOLD SCOPE${tail}`; + expect(decide(f)?.option).toBe('HOLD SCOPE'); + }); + test(`matching later explanation preserves the current decision: ${JSON.stringify(tail)}`, () => { + const f = clone749(); declaration749(f).text += `\n\nMode: HOLD SCOPE${tail}`; + expect(decide(f)?.option).toBe('HOLD SCOPE'); + }); + test(`conflicting later explanation withdraws the current decision: ${JSON.stringify(tail)}`, () => { + const f = clone749(); declaration749(f).text += `\n\nMode: SCOPE EXPANSION${tail}`; + expect(decide(f)).toBeNull(); + }); +} + +const incompleteFields = [ + 'Mode: HOLD SCOPE (saved preference; recorded.', + 'Mode: HOLD SCOPE (saved preference (recorded).', + 'Mode: HOLD SCOPE (saved preference)). Recorded.', + 'Mode: HOLD SCOPE (saved preference)SCOPE EXPANSION', + 'Mode: HOLD SCOPE. A following explanation (unfinished.', + 'Mode: HOLD SCOPE (saved preference). If approved.', + 'Mode: HOLD SCOPE (saved preference (if approved)). Recorded.', + 'Mode: HOLD SCOPE (saved preference). Not yet selected.', + 'Mode: HOLD SCOPE (saved preference). I did not auto-decide the review mode.', + 'Mode: HOLD SCOPE (saved preference). This decision is withdrawn.', + 'Mode decision pending: HOLD SCOPE (saved preference). Recorded.', + 'Mode decision tentative: HOLD SCOPE (saved preference). Recorded.', + 'Mode: CUSTOM MODE (saved preference). Recorded.', + 'Mode: HOLD SCOPE / SCOPE EXPANSION (saved preference). Recorded.', + 'Mode: HOLD SCOPE (saved preference).\nMode decision pending: HOLD SCOPE', +]; +for (const field of incompleteFields) { + test(`explanatory prose cannot complete an unsupported field: ${JSON.stringify(field)}`, () => { + const f = clone749(); declaration749(f).text = field; + expect(decide(f)).toBeNull(); + }); + test(`later unsupported field retracts the earlier decision: ${JSON.stringify(field)}`, () => { + const f = clone749(); declaration749(f).text += `\n\n${field}`; + expect(decide(f)).toBeNull(); + }); +} + +for (const [name, mutate] of Object.entries({ + 'missing owned append': (f: any) => { f.options.stateEvidence.records = []; }, + 'foreign owned session': (f: any) => { f.options.stateEvidence.records[0].session_id = 'foreign'; }, + 'duplicate owned append': (f: any) => { f.options.stateEvidence.records.push({ ...f.options.stateEvidence.records[0] }); }, + 'conflicting logged choice': (f: any) => { f.options.stateEvidence.records[0].user_choice = 'SCOPE EXPANSION'; }, + 'failed preamble': (f: any) => { + const preamble = f.tools.find((e: any) => e.kind === 'use' && e.input?.command?.includes('gstack-skill-start')); + f.tools.find((e: any) => e.kind === 'result' && e.toolUseId === preamble.toolUseId).isError = true; + }, + 'native question': (f: any) => { f.transcript.calls.push({ sessionId: f.options.sessionId }); }, + 'declaration before owned append': (f: any) => { declaration749(f).timestamp = '2026-09-16T12:12:00.000Z'; }, + 'quoted declaration': (f: any) => { declaration749(f).text = '> Mode: HOLD SCOPE (saved preference). Recorded.'; }, + 'example declaration': (f: any) => { declaration749(f).text = 'Example:\nMode: HOLD SCOPE (saved preference). Recorded.'; }, +})) test(`captured explanatory mode still requires ${name}`, () => { + const f = clone749(); mutate(f); expect(decide(f)).toBeNull(); +}); + +const reviewModes = ['HOLD SCOPE', 'SCOPE EXPANSION', 'SELECTIVE EXPANSION', 'SCOPE REDUCTION']; +for (const mode of reviewModes) { + for (const tail of [' (saved preference). Recorded for this invocation.', + ' (saved preference (confirmed); recorded). No further mode decision.', + ` (saved preference). ${mode} is recorded for this invocation.`]) { + test(`each complete review mode supports an unambiguous explanatory suffix: ${mode}${tail}`, () => { + const f = clone749(); + Object.assign(f.options.stateEvidence.records[0], { user_choice: mode, recommended: mode }); + declaration749(f).text = `Mode: ${mode}${tail}`; + expect(decide(f)?.option).toBe(mode); + }); + } + for (const other of reviewModes.filter(value => value !== mode)) { + for (const connector of [' or ', ' versus ', ' vs. ', ' / ', ' | ', '; or ', ', choose ', + '. Alternatively, select ', ' — instead choose ', ' (otherwise choose ', ' rather than ']) { + test(`a second distinct mode in the suffix stays ambiguous: ${mode}${connector}${other}`, () => { + const f = clone749(); + Object.assign(f.options.stateEvidence.records[0], { user_choice: mode, recommended: mode }); + const tail = connector.startsWith(' (') ? ')' : ''; + declaration749(f).text = `Mode: ${mode} (saved preference)${connector}${other}${tail}`; + expect(decide(f)).toBeNull(); + }); + } + } +} + +test('alternate current mode spellings remain ambiguous after an explanatory parenthetical', () => { + for (const alternative of ['scope expansion', 'SCOPE_EXPANSION', 'SCOPE EXPANSION']) { + const f = clone749(); declaration749(f).text = `Mode: HOLD SCOPE (saved preference); ${alternative}`; + expect(decide(f)).toBeNull(); + } +}); + +test('generic annotation vocabulary retains its original parenthetical boundaries', () => { + for (const suffix of ['', ' or Startup', '; or Startup', ' versus Startup', '. Recorded for this invocation.']) { + const f = structuredClone(annotations.attempts[0]) as any; + f.options.skillName = 'office-hours'; + f.tools.find((e: any) => e.kind === 'use' && e.name === 'Skill').input.skill = 'office-hours'; + const message = f.transcript.assistantMessages.find((m: any) => m.text.startsWith('Auto-decided')); + message.text = `Auto-decided workflow → **Builder** (your preference). Change with /plan-tune.\n\nMode: Builder (saved preference)${suffix}`; + expect(decide(f)?.option ?? null).toBe(suffix ? null : 'Builder'); + } +}); +}); + +describe('auto-decide-recommendation-scope', () => { +const capture = capture_auto_decide_recommendation_scope; +const clone = () => structuredClone(capture) as any; +const decide = (f: any) => findNativeAutoDecision(f.transcript, f.tools, f.options); +const message = (f: any) => f.transcript.assistantMessages.find((m: any) => + m.timestamp === '2026-09-17T02:23:11.495Z'); + +test('actual completed mode and recommendation commentary match the retained owned audit', () => { + const f = clone(), result = decide(f); + expect(result?.option).toBe('HOLD SCOPE'); + expect(result?.annotation).toBe(message(f).text); + expect(result?.stateRecord).toEqual(f.options.stateEvidence.records[0]); + expect(result?.preambleToolUseId).toBe('toolu_01Ni4b9NZeiexmcAz1jRTUa4'); +}); + +const modes = ['HOLD SCOPE', 'SCOPE EXPANSION', 'SELECTIVE EXPANSION', 'SCOPE REDUCTION']; +for (const mode of modes) for (const commentary of [ + 'recommendation would have been the same', + 'my recommendation might differ without the saved preference', + `the recommendation would still be ${mode}`, + 'our recommendation will remain unchanged', + 'recommendation stays the same unless the product context changes', +]) test(`completed ${mode} is separate from ${commentary}`, () => { + const f = clone(); + Object.assign(f.options.stateEvidence.records[0], { user_choice: mode, recommended: mode }); + message(f).text = `Decision: review mode is ${mode} (${commentary}).`; + expect(decide(f)?.option).toBe(mode); +}); + +for (const separator of ['; ', ', ', '. ', ' — ', ' – ', ' - ', ' (']) { + test(`recommendation assertion has an explicit boundary: ${JSON.stringify(separator)}`, () => { + const f = clone(); + message(f).text = `Mode: HOLD SCOPE${separator}recommendation would have been unchanged${separator === ' (' ? ')' : ''}.`; + expect(decide(f)?.option).toBe('HOLD SCOPE'); + }); +} + +const uncertain = [ + 'Mode: would choose HOLD SCOPE.', + 'Mode: HOLD SCOPE if approved.', + 'Mode: HOLD SCOPE unless you object.', + 'Mode: HOLD SCOPE (I will make this selection).', + 'Mode: HOLD SCOPE (this choice might change).', + 'Mode: HOLD SCOPE (recommendation would be the same; if approved).', + 'Mode: HOLD SCOPE (recommendation would be the same, unless you object).', + 'Mode: HOLD SCOPE (recommendation would be the same and I will select it later).', + 'Mode: HOLD SCOPE (recommendation would be the same but the choice might change).', + 'Mode: HOLD SCOPE (recommendation would be the same while we would still need approval).', + 'Mode: HOLD SCOPE (recommendation would be the same; selection is pending).', + 'Mode: HOLD SCOPE (recommendation says the decision would be conditional).', + 'Mode: HOLD SCOPE (recommendation would still be SCOPE EXPANSION).', + 'Mode: HOLD SCOPE (recommendation would be unchanged). Not yet selected.', + 'Mode: HOLD SCOPE (recommendation would be unchanged). This decision is withdrawn.', + 'Mode pending: HOLD SCOPE (recommendation would be unchanged).', + 'Mode: not HOLD SCOPE (recommendation would be unchanged).', + 'Mode: HOLD SCOPE for a future review (recommendation would be unchanged).', + 'Mode: HOLD SCOPE for another draft (recommendation would be unchanged).', + 'Mode: HOLD SCOPELESS (recommendation would be unchanged).', + 'Mode: HOLD SCOPE (recommendation would be unchanged.', +]; +for (const text of uncertain) { + test(`commentary does not authenticate an uncertain choice: ${text}`, () => { + const f = clone(); message(f).text = text; + expect(decide(f)).toBeNull(); + }); + test(`later uncertain choice retracts the original completed decision: ${text}`, () => { + const f = clone(); message(f).text += `\n\nCorrection: ${text}`; + expect(decide(f)).toBeNull(); + }); +} + +for (const [name, wrap] of [ + ['quoted', (s: string) => `> ${s}`], + ['indented', (s: string) => ` ${s}`], + ['fenced', (s: string) => `\`\`\`text\n${s}\n\`\`\``], + ['historical', (s: string) => `Previous decision:\n${s}`], + ['example', (s: string) => `Example:\n${s}`], +] as const) test(`recommendation commentary cannot authenticate ${name} declarations`, () => { + const f = clone(); message(f).text = wrap('Mode: HOLD SCOPE (recommendation would be unchanged).'); + expect(decide(f)).toBeNull(); +}); + +for (const [name, mutate] of Object.entries({ + 'missing owned log': (f: any) => { f.options.stateEvidence.records = []; }, + 'duplicate owned log': (f: any) => { f.options.stateEvidence.records.push({ ...f.options.stateEvidence.records[0] }); }, + 'foreign audit session': (f: any) => { f.options.stateEvidence.records[0].session_id = 'foreign'; }, + 'different selected choice': (f: any) => { f.options.stateEvidence.records[0].user_choice = 'SCOPE EXPANSION'; }, + 'different recommendation': (f: any) => { f.options.stateEvidence.records[0].recommended = 'SCOPE EXPANSION'; }, + 'nonautomatic record': (f: any) => { f.options.stateEvidence.records[0].auto_decided = false; }, + 'foreign native session': (f: any) => { f.options.sessionId = 'foreign'; }, + 'missing successful preamble': (f: any) => { f.tools = f.tools.filter((e: any) => e.toolUseId !== 'toolu_01Ni4b9NZeiexmcAz1jRTUa4'); }, + 'future audit': (f: any) => { f.options.stateEvidence.records[0].ts = new Date(f.options.now + 1).toISOString(); }, + 'decision before completed log': (f: any) => { message(f).timestamp = new Date(Date.parse(f.options.stateEvidence.records[0].ts) - 1).toISOString(); }, + 'surfaced native question': (f: any) => { f.transcript.calls.push({ sessionId: f.options.sessionId }); }, + 'surfaced prose question': (f: any) => { f.options.proseQuestionObserved = true; }, +})) test(`actual recommendation commentary retains ${name} rejection`, () => { + const f = clone(); mutate(f); expect(decide(f)).toBeNull(); +}); +}); + +describe('auto-decide-saved-ai', () => { +const captured = captured_auto_decide_saved_ai; +const retry = retry_auto_decide_saved_ai; +const clone=()=>structuredClone(captured) as any; +const decision=(f=clone())=>findNativeAutoDecision(f.transcript,f.tools,f.options); +const message=(f:any)=>f.transcript.assistantMessages.find((m:any)=>m.text.includes('Auto-decided')); + +test('actual saved mode preference annotation is a completed native auto-decision',()=>{ + const f=clone(), result=decision(f); + expect(result).not.toBeNull(); + expect(result!.option).toBe('HOLD SCOPE'); + expect(message(f).text).toContain(result!.annotation); + expect(f.transcript.calls).toEqual([]); +}); + +test('saved preference is bound to this invoked skill and an agreeing current mode',()=>{ + for(const change of [ + (s:string)=>s.replace('`plan-ceo-review-mode`','`plan-design-review-mode`'), + (s:string)=>s.replace('`plan-ceo-review-mode`','`plan-ceo-review-routing`'), + (s:string)=>s.replace('**Review mode: HOLD SCOPE.**','**Review mode: SCOPE EXPANSION.**'), + (s:string)=>s.replace('**Review mode: HOLD SCOPE.**\n\n',''), + (s:string)=>s.replace('"Select review mode"','"Select report folder"'), + (s:string)=>s.replace('(your saved preference on','(a proposed preference on'), + (s:string)=>s.replace('Change with /plan-tune.',''), + (s:string)=>s.replace('Auto-decided','I will auto-decide'), + ]) {const f=clone();message(f).text=change(message(f).text);expect(decision(f)).toBeNull();} +}); + +test('prefixed examples, quotations and hypothetical notices do not assert a current choice',()=>{ + for(const change of [ + (s:string)=>'Example:\n\n'+s, + (s:string)=>'```text\n'+s+'\n```', + (s:string)=>s.split('\n').map(l=>'> '+l).join('\n'), + (s:string)=>s.replace('Heads-up from the preamble: unshipped work on this branch','Heads-up from the preamble: a hypothetical example'), + (s:string)=>s.replace('Auto-decided "Select',' Auto-decided "Select'), + (s:string)=>s.replace('Auto-decided "Select','If approved, Auto-decided "Select'), + ]) {const f=clone();message(f).text=change(message(f).text);expect(decision(f)).toBeNull();} +}); + +test('failed loads, foreign sessions, actual questions and later withdrawals retain precedence',()=>{ + for(const mutate of [ + (f:any)=>{f.options.sessionId='foreign';}, + (f:any)=>{const use=f.tools.find((t:any)=>t.kind==='use'&&t.name==='Skill');f.tools.find((t:any)=>t.kind==='result'&&t.toolUseId===use.toolUseId).isError=true;}, + (f:any)=>{f.transcript.calls.push({sessionId:f.options.sessionId,toolUseId:'actual-question'});}, + (f:any)=>{f.options.now=Date.parse(message(f).timestamp)-1;}, + (f:any)=>{message(f).text+='\n\nCorrection: I withdraw this decision.';}, + (f:any)=>{message(f).text+='\n\n**Review mode: SCOPE EXPANSION.**';}, + ]) {const f=clone();mutate(f);expect(decision(f)).toBeNull();} +}); +test('actual retry mode-decision heading retains its own annotation, excluding prior foreign text',()=>{ + const f:any=structuredClone(retry), actual=decision(f); + expect(actual).not.toBeNull();expect(actual!.sessionId).toBe(f.options.sessionId); + expect(actual!.option).toBe('HOLD SCOPE'); + expect(actual!.annotation).toContain('(your preference)'); + const own=f.transcript.assistantMessages.filter((m:any)=>m.sessionId===f.options.sessionId); + f.transcript.assistantMessages=f.transcript.assistantMessages.filter((m:any)=>m.sessionId!==f.options.sessionId); + expect(decision(f)).toBeNull();expect(own.length).toBeGreaterThan(0); +}); + +test('retry heading cannot supply a hypothetical, different decision, or withdrawn selection',()=>{ + for(const change of [ + (s:string)=>'Example:\n\n'+s, + (s:string)=>s.replace('Review mode for the deterministic','Review mode for the hypothetical'), + (s:string)=>s.replace('D1 — Review mode','D1 — Report destination'), + (s:string)=>s.replace('Auto-decided "Review mode:', 'Auto-decided "Report destination:'), + (s:string)=>s.replace('→ **HOLD SCOPE**','→ **Save a file**'), + (s:string)=>s+'\n\nCorrection: I withdraw this selection.', + (s:string)=>s+'\n\n**Review mode: SCOPE EXPANSION.**', + (s:string)=>s.replace('Heads-up from gstack: there is unshipped work on this branch','Heads-up from gstack: here is an example'), + ]) {const f:any=structuredClone(retry),m=f.transcript.assistantMessages.find((m:any)=>m.sessionId===f.options.sessionId&&m.text.includes('Auto-decided'));m.text=change(m.text);expect(decision(f)).toBeNull();} +}); +}); + +describe('auto-decide-structured', () => { +const capture = capture_auto_decide_structured; +const priorAnnotation = captured_auto_decide_saved_ai; +const completedModeCapture = completedModeCapture_auto_decide_structured; +const statusFixture = completedModeCapture_auto_decide_structured; +const clone = () => structuredClone(capture) as any; +const decision = (f = clone()) => findNativeAutoDecision(f.transcript, f.tools, f.options); +test('actual slash expansion with completed preference log and current mode is an auto-decision', () => { + const f = clone(); + expect(f.tools.some((e: any) => e.name === 'Skill')).toBe(false); + expect(f.transcript.calls).toEqual([]); + const result = decision(f); + expect(result).not.toBeNull(); + expect(result!.option).toBe('HOLD SCOPE'); +}); + +const use = (f: any, name: string) => f.tools.find((e: any) => e.kind === 'use' && e.input?.command?.includes(name)); +const ack = (f: any, request: any) => f.tools.find((e: any) => e.kind === 'result' && e.toolUseId === request.toolUseId); +const modeMessage = (f: any) => f.transcript.assistantMessages.find((m: any) => m.text.includes('**Mode:')); +const changeLog = (f: any, modify: (log: any) => void) => { + const request = use(f, 'gstack-question-log'), match = /'(\{.*\})'/.exec(request.input.command)!; + const value = JSON.parse(match[1]!); modify(value); + request.input.command = request.input.command.replace(match[1], JSON.stringify(value)); +}; + +for (const [label, mutate] of Object.entries({ + 'missing transcript': (f: any) => { f.transcript.status = 'missing'; }, + 'foreign owned session': (f: any) => { f.options.sessionId = 'foreign'; }, + 'wrong invoked skill': (f: any) => { f.options.skillName = 'plan-eng-review'; }, + 'pre-command evidence': (f: any) => { f.options.commandStartedAt = Date.parse(modeMessage(f).timestamp); }, + 'future final statement': (f: any) => { f.options.now = Date.parse(modeMessage(f).timestamp) - 1; }, + 'invalid final timestamp': (f: any) => { modeMessage(f).timestamp = 'invalid'; }, + 'native question': (f: any) => { f.transcript.calls.push({ sessionId: f.options.sessionId, toolUseId: 'asked' }); }, + 'malformed native question tool': (f: any) => { f.tools.push({ ...use(f, 'gstack-question-log'), toolUseId: 'asked', name: 'mcp__ask__AskUserQuestion', input: {} }); }, + 'earlier visible prose question': (f: any) => { f.options.proseQuestionObserved = true; }, + 'public reply request': (f: any) => { modeMessage(f).text += '\nReply with A or B.'; }, + 'public option list': (f: any) => { modeMessage(f).text += '\nA) Hold scope\nB) Expand scope'; }, + 'no preamble': (f: any) => { const request = use(f, 'gstack-skill-start'); f.tools = f.tools.filter((e: any) => e.toolUseId !== request.toolUseId); }, + 'preamble failed': (f: any) => { ack(f, use(f, 'gstack-skill-start')).isError = true; }, + 'preamble missing ACK': (f: any) => { const request = use(f, 'gstack-skill-start'); f.tools = f.tools.filter((e: any) => e !== ack(f, request)); }, + 'preamble duplicate': (f: any) => { f.tools.push({ ...use(f, 'gstack-skill-start') }); }, + 'wrong preamble skill': (f: any) => { use(f, 'gstack-skill-start').input.command = use(f, 'gstack-skill-start').input.command.replace('--skill "plan-ceo-review"', '--skill "plan-eng-review"'); }, + 'question tuning disabled': (f: any) => { const result = ack(f, use(f, 'gstack-skill-start')); result.content = result.content.replace('QUESTION_TUNING: true', 'QUESTION_TUNING: false'); }, + 'ambiguous preamble session': (f: any) => { ack(f, use(f, 'gstack-skill-start')).content = 'SKILL_START_PROTO: 1\nQUESTION_TUNING: true\nSESSION_ID: duplicate\n' + ack(f, use(f, 'gstack-skill-start')).content; }, + 'nonzero preference': (f: any) => { ack(f, use(f, 'gstack-question-preference')).content = 'AUTO_DECIDE\nEXIT: 1'; }, + 'ASK preference': (f: any) => { ack(f, use(f, 'gstack-question-preference')).content = 'ASK\nEXIT: 0'; }, + 'preference error': (f: any) => { ack(f, use(f, 'gstack-question-preference')).isError = true; }, + 'wrong preference id': (f: any) => { use(f, 'gstack-question-preference').input.command = use(f, 'gstack-question-preference').input.command.replace('--check "plan-ceo-review-mode"', '--check "plan-ceo-review-other"'); }, + 'no preference check': (f: any) => { const request = use(f, 'gstack-question-preference'); f.tools = f.tools.filter((e: any) => e.toolUseId !== request.toolUseId); }, + 'unacknowledged log': (f: any) => { const request = use(f, 'gstack-question-log'); f.tools = f.tools.filter((e: any) => e !== ack(f, request)); }, + 'failed log': (f: any) => { ack(f, use(f, 'gstack-question-log')).isError = true; }, + 'fallback log result': (f: any) => { ack(f, use(f, 'gstack-question-log')).content = 'log unavailable (best-effort)'; }, + 'wrong log session': (f: any) => changeLog(f, log => { log.session_id = 'foreign'; }), + 'wrong log skill': (f: any) => changeLog(f, log => { log.skill = 'plan-eng-review'; }), + 'wrong log question id': (f: any) => changeLog(f, log => { log.question_id = 'plan-ceo-review-scope'; }), + 'nonautomatic log': (f: any) => changeLog(f, log => { log.auto_decided = false; }), + 'string automatic flag': (f: any) => changeLog(f, log => { log.auto_decided = 'true'; }), + 'unmatched recommendation': (f: any) => changeLog(f, log => { log.recommended = 'SCOPE_EXPANSION'; }), + 'different logged mode': (f: any) => changeLog(f, log => { log.recommended = log.user_choice = 'SCOPE_EXPANSION'; }), + 'arbitrary logged value': (f: any) => changeLog(f, log => { log.recommended = log.user_choice = 'APPROVE_SCOPE'; }), + 'nondecision summary': (f: any) => changeLog(f, log => { log.question_summary = ''; }), + 'later checked preference': (f: any) => { ack(f, use(f, 'gstack-question-preference')).timestamp = modeMessage(f).timestamp; }, + 'mode before log ACK': (f: any) => { modeMessage(f).timestamp = use(f, 'gstack-question-log').timestamp; }, + 'reversed log ACK': (f: any) => { ack(f, use(f, 'gstack-question-log')).timestamp = use(f, 'gstack-question-preference').timestamp; }, + 'duplicate log ACK': (f: any) => { f.tools.push({ ...ack(f, use(f, 'gstack-question-log')) }); }, + 'foreign log ACK': (f: any) => { ack(f, use(f, 'gstack-question-log')).sessionId = 'foreign'; }, + 'missing current statement': (f: any) => { modeMessage(f).text = 'Done. Waiting for your next instruction.'; }, +})) test(`structured current mode rejects ${label}`, () => { + const f = clone(); mutate(f); expect(decision(f)).toBeNull(); +}); + +for (const name of ['gstack-skill-start', 'gstack-question-preference', 'gstack-question-log']) { + for (const [label, change] of Object.entries({ + 'echoed source': (s: string) => `echo '${s.replaceAll("'", "'\\''")}'`, + 'conditional command': (s: string) => `false && ${s}`, + 'commented source': (s: string) => `# ${s}`, + 'extra prefix command': (s: string) => `true; ${s}`, + 'extra suffix command': (s: string) => `${s}; true`, + 'command substitution': (s: string) => `echo "$(${s})"`, + })) test(`${name} cannot authenticate ${label}`, () => { + const f = clone(); use(f, name).input.command = change(use(f, name).input.command); expect(decision(f)).toBeNull(); + }); +} + +for (const [label, text] of Object.entries({ + 'plain current field': 'Mode: HOLD SCOPE.', + 'parenthetical explanation with punctuation': 'Mode: HOLD SCOPE (saved preference, confirmed).', + 'parenthetical review explanation': '**Review mode: HOLD SCOPE (saved preference; confirmed).**', + 'current review field': '**Review mode: HOLD SCOPE.**', + 'compact completion': '**STATUS: DONE**\n\nMode: HOLD SCOPE', + 'bullet conclusion': 'The requested routing decision is complete.\n\n- **Mode: HOLD SCOPE**, using the saved preference.\n\nThe substantive review is deferred.', + 'quoted historical contradiction': 'Mode: HOLD SCOPE.\n\nEarlier example: "Review mode: SCOPE EXPANSION."', +})) test(`completed structured log supports ${label} without exact annotation prose`, () => { + const f = clone(); modeMessage(f).text = text; + const result = decision(f); expect(result?.option).toBe('HOLD SCOPE'); + expect(result?.skillToolUseId).toBeUndefined(); + expect(result?.preambleToolUseId).toBe(use(f, 'gstack-skill-start').toolUseId); + expect(result?.annotation).toBe(text); +}); + +for (const text of [ + '> Mode: HOLD SCOPE.', ' Mode: HOLD SCOPE.', '`Mode: HOLD SCOPE.`', + '```text\nMode: HOLD SCOPE.\n```', 'Example:\n\nMode: HOLD SCOPE.', + 'Previous transcript:\n\nMode: HOLD SCOPE.', 'If approved, Mode: HOLD SCOPE.', + 'Mode: HOLD SCOPE, if you approve.', 'Mode: HOLD SCOPE, pending approval.', + 'Mode: HOLD SCOPE?', 'Mode: HOLD SCOPELESS.', + 'Mode: HOLD SCOPE (withdrawn).', 'Mode: HOLD SCOPE (retracted).', + 'Mode: HOLD SCOPE.\n\nMode: HOLD SCOPE (pending approval).', + 'Mode: HOLD SCOPE.\n\nCorrection: I withdraw this decision.', + 'Mode: HOLD SCOPE.\n\nI did not auto-decide the review mode.', + 'Mode: HOLD SCOPE.\n\nCorrection: Mode: SCOPE EXPANSION.', + 'Mode: HOLD SCOPE.\n\nMode: SCOPE EXPANSION.', +]) test(`quoted, conditional or withdrawn mode has no completed choice: ${JSON.stringify(text)}`, () => { + const f = clone(); modeMessage(f).text = text; expect(decision(f)).toBeNull(); +}); + +test('the same command contracts also support direct literal invocations and quiet ACKs', () => { + const f = clone(); + use(f, 'gstack-skill-start').input.command = '"$HOME/.claude/skills/gstack/bin/gstack-skill-start" --model claude --skill plan-ceo-review --parent-pid "$PPID"'; + use(f, 'gstack-question-preference').input.command = '~/.claude/skills/gstack/bin/gstack-question-preference --check plan-ceo-review-mode'; + ack(f, use(f, 'gstack-question-preference')).content = 'AUTO_DECIDE\n'; + use(f, 'gstack-question-log').input.command = use(f, 'gstack-question-log').input.command.split(' 2>/dev/null')[0]; + ack(f, use(f, 'gstack-question-log')).content = ''; + expect(decision(f)?.option).toBe('HOLD SCOPE'); +}); +for (const status of ['undecided', 'not selected', 'pending approval', 'none']) { + test(`later Review mode: ${status} withdraws both existing annotation and structured decision`, () => { + const previous: any = structuredClone(priorAnnotation); + previous.transcript.assistantMessages.find((m: any) => m.text.includes('Auto-decided')).text += `\n\nReview mode: ${status}.`; + expect(findNativeAutoDecision(previous.transcript, previous.tools, previous.options)).toBeNull(); + const f = clone(); modeMessage(f).text += `\n\nReview mode: ${status}.`; + expect(decision(f)).toBeNull(); + }); + test(`later Mode: ${status} withdraws a structured decision`, () => { + const f = clone(); modeMessage(f).text += `\n\n- **Mode: ${status}.**`; + expect(decision(f)).toBeNull(); + }); +} + +for (const name of ['gstack-question-preference', 'gstack-question-log']) test(`${name} cannot borrow an earlier success after a contradictory current call`, () => { + const f = clone(), request = structuredClone(use(f, name)), result = structuredClone(ack(f, request)); + request.toolUseId += '-later'; result.toolUseId = request.toolUseId; + request.timestamp = result.timestamp = new Date(Date.parse(modeMessage(f).timestamp) - 1).toISOString(); + if (name === 'gstack-question-preference') result.content = 'ASK\nEXIT: 0'; + else request.input.command = request.input.command.replace('"auto_decided":true', '"auto_decided":false'); + f.tools.push(request, result); expect(decision(f)).toBeNull(); +}); + +test('a literal command cannot treat a physical newline as argument whitespace', () => { + const f = clone(); + use(f, 'gstack-question-log').input.command = use(f, 'gstack-question-log').input.command.replace("gstack-question-log '", "gstack-question-log\n'"); + expect(decision(f)).toBeNull(); +}); + +for (const fallback of ['"LOGGED"', '" LOGGED "', '"\\x4cOGGED"', '-e "\\x4cOGGED"']) + test(`a failure branch cannot impersonate the question-log success marker: ${fallback}`, () => { + const f = clone(), request = use(f, 'gstack-question-log'); + request.input.command = request.input.command.replace('"log unavailable (best-effort)"', fallback); + expect(decision(f)).toBeNull(); + }); +{ +const copy=()=>structuredClone(completedModeCapture); +const check=(f:any)=>findNativeAutoDecision(f.transcript,f.tools,f.options); +const message=(f:any)=>f.transcript.assistantMessages.find((m:any)=>m.text.includes('Mode decision done:')); +const logUse=(f:any)=>f.tools.find((t:any)=>t.kind==='use'&&t.input?.command?.includes('gstack-question-log')); +test('actual owned public attempt fails original and completes mode-only with full acknowledged authority',()=>{ + const f=copy();const v=check(f);expect(v?.option).toBe('HOLD SCOPE');expect(v?.questionLogToolUseId).toBe(logUse(f).toolUseId); +}); +const mutations:Recordvoid>={ + 'unlogged':f=>{const id=logUse(f).toolUseId;f.tools=f.tools.filter((t:any)=>t.toolUseId!==id)}, + 'failed log':f=>{f.tools.find((t:any)=>t.kind==='result'&&t.toolUseId===logUse(f).toolUseId).isError=true}, + 'masked log failure':f=>{logUse(f).input.command=logUse(f).input.command.replace('&& echo','; echo')}, + 'wrong returned marker':f=>{f.tools.find((t:any)=>t.kind==='result'&&t.toolUseId===logUse(f).toolUseId).content='LOG_FAILED (best-effort)'}, + 'unmatched quote':f=>{logUse(f).input.command=logUse(f).input.command.replace('"LOGGED"','"LOGGED')}, + 'foreign session':f=>{f.options.sessionId='foreign'}, + 'wrong mode':f=>{message(f).text=message(f).text.replace('done: HOLD SCOPE','done: SCOPE EXPANSION')}, + 'unfinished':f=>{message(f).text=message(f).text.replace('Mode decision done:','Mode decision pending:')}, + 'late declaration':f=>{message(f).timestamp=new Date(f.options.now+1000).toISOString()}, + 'prior declaration':f=>{message(f).timestamp=new Date(f.options.commandStartedAt-1000).toISOString()}, + 'cancelled':f=>{message(f).text+='\n\nI cancel this decision.'}, + 'wrong later completed mode':f=>{message(f).text+='\n\nMode decision done: SCOPE EXPANSION'}, + 'quoted declaration':f=>{message(f).text='> '+message(f).text}, + 'hypothetical':f=>{message(f).text='Example:\n'+message(f).text}, + 'conditional':f=>{message(f).text=message(f).text.replace('done: HOLD SCOPE','done: HOLD SCOPE (if approved)')}, + 'native question surfaced':f=>{f.transcript.calls.push({sessionId:f.options.sessionId})}, + 'wrong logged mode':f=>{logUse(f).input.command=logUse(f).input.command.replace('"user_choice":"HOLD SCOPE"','"user_choice":"SCOPE EXPANSION"')}, +}; +for(const [name,mutate] of Object.entries(mutations))test(name,()=>{const f=copy();mutate(f);expect(check(f)).toBeNull()}); + +for(const completion of ['done','complete','completed']) { + test(`completed mode class ${completion}`,()=>{const f=copy();message(f).text=message(f).text.replace('decision done:','decision '+completion+':');expect(check(f)?.option).toBe('HOLD SCOPE')}); + test(`conflicting later completed mode ${completion}`,()=>{const f=copy();message(f).text+='\n\nMode decision '+completion+': SCOPE EXPANSION';expect(check(f)).toBeNull()}); + test(`unfinished completed mode ${completion}`,()=>{const f=copy();message(f).text=message(f).text.replace('done: HOLD SCOPE',completion+': HOLD SCOPE (pending approval)');expect(check(f)).toBeNull()}); +} +test('paired single-quoted success token retains exact shell ACK',()=>{const f=copy();logUse(f).input.command=logUse(f).input.command.replace('"LOGGED"',"'LOGGED'");expect(check(f)?.option).toBe('HOLD SCOPE')}); +test('unpaired single-quoted success token cannot authenticate log',()=>{const f=copy();logUse(f).input.command=logUse(f).input.command.replace('"LOGGED"',"'LOGGED");expect(check(f)).toBeNull()}); + +} +{ +const fixture=statusFixture; +const fixed=findNativeAutoDecision; +const copy=()=>structuredClone(fixture) as any; +const message=(f:any)=>f.transcript.assistantMessages.find((m:any)=>m.text.includes('Mode decision done:')); +const check=(f:any)=>fixed(f.transcript,f.tools,f.options); +test('current pending status retracts the completed owned mode',()=>{const f=copy();message(f).text+='\n\nMode decision pending: HOLD SCOPE';expect(check(f)).toBeNull()}); +for(const status of ['pending','pending approval','unfinished','incomplete','cancelled','canceled','withdrawn','retracted','revoked','undecided','proposed','not selected','not decided','not yet complete','in progress','on hold','unknown']){ + test(`unfinished declaration ${status}`,()=>{const f=copy();message(f).text=message(f).text.replace('decision done:','decision '+status+':');expect(check(f)).toBeNull()}); + test(`later unfinished status ${status}`,()=>{const f=copy();message(f).text+='\n\nMode decision '+status+': HOLD SCOPE';expect(check(f)).toBeNull()}); + test(`quoted historical status ${status}`,()=>{const f=copy();message(f).text+='\n\n> Historical example:\n> Mode decision '+status+': HOLD SCOPE';expect(check(f)?.option).toBe('HOLD SCOPE')}); +} +for(const status of ['done','complete','completed']){ + test(`same current completed field ${status}`,()=>{const f=copy();message(f).text+='\n\nMode decision '+status+': HOLD SCOPE';expect(check(f)?.option).toBe('HOLD SCOPE')}); + test(`completed conflicting field ${status}`,()=>{const f=copy();message(f).text+='\n\nMode decision '+status+': SCOPE EXPANSION';expect(check(f)).toBeNull()}); +} +for(const status of ['unfinished','incomplete','pending approval','cancelled','not completed'])test(`unfinished value suffix ${status}`,()=>{const f=copy();message(f).text+='\n\nMode decision done: HOLD SCOPE ('+status+')';expect(check(f)).toBeNull()}); +for(const text of ['Historical example: Mode decision pending: HOLD SCOPE','```\nMode decision pending: HOLD SCOPE\n```','"Mode decision cancelled: HOLD SCOPE"'])test(`unasserted historical field ${text}`,()=>{const f=copy();message(f).text+='\n\n'+text;expect(check(f)?.option).toBe('HOLD SCOPE')}); +} +}); + +describe('auto-decide-target-identity', () => { +const capture = capture_auto_decide_target_identity; +const clone = () => structuredClone(capture) as any; +const message = (f: any) => f.transcript.assistantMessages.at(-1); +const decide = (f: any) => findNativeAutoDecision(f.transcript, f.tools, f.options); + +test('actual quoted current title and completed owned audit produce the original mode decision', () => { + const f = clone(), result = decide(f); + expect(result?.option).toBe('HOLD SCOPE'); + expect(result?.annotation).toBe(message(f).text); + expect(result?.stateRecord).toEqual(f.options.stateEvidence.records[0]); + expect(result?.preambleToolUseId).toBe('toolu_01KbsH6ybJxbNozwbSXywVbb'); +}); + +const title = 'deterministic skill-list ordering'; +const modes = ['HOLD SCOPE', 'SCOPE EXPANSION', 'SELECTIVE EXPANSION', 'SCOPE REDUCTION']; +for (const mode of modes) for (const quote of [(s: string) => `"${s}"`, (s: string) => `“${s}”`, (s: string) => `\`${s}\``]) { + for (const wrapper of ['', ' draft', ' plan']) test(`${mode} quoted title agrees with one audit wrapper: ${quote(title)}${wrapper}`, () => { + const f = clone(); + Object.assign(f.options.stateEvidence.records[0], { user_choice: mode, recommended: mode, question_summary: `Select review mode for ${title}${wrapper}` }); + message(f).text = `Decision: ${mode} for ${quote(title)}.\n\nMode: ${mode}, auto-selected using the saved preference.`; + expect(decide(f)?.option).toBe(mode); + expect(decide(f)?.annotation).toBe(message(f).text); + }); +} +for (const [declared, recorded] of [ + [`"${title}" draft`, `"${title}"`], + [`"${title}" plan`, `${title} draft`], + [title, `${title} draft`], + [`${title} draft`, title], + ['"release plan"', 'release plan draft'], + ['"what if ordering"', 'what if ordering draft'], + ['"ordering v2. current"', '"ordering v2. current" draft'], +]) test(`exact title identity with syntactic wrapper: ${declared} / ${recorded}`, () => { + const f = clone(); f.options.stateEvidence.records[0].question_summary = `Select mode for ${recorded}`; + message(f).text = `Decision: HOLD SCOPE for ${declared}.`; + expect(decide(f)?.option).toBe('HOLD SCOPE'); +}); + +test('quoted target and mode labels remain case insensitive', () => { + const f = clone(); message(f).text = `decision: hold scope FOR "${title}".`; + expect(decide(f)?.option).toBe('HOLD SCOPE'); +}); + +const negatives: Array<[string, string]> = [ + ['"deterministic skill-list sorting"', `${title} draft`], + ['"skill-list ordering"', `${title} draft`], + [`"${title}-v2"`, `${title} draft`], + [`"${title} extra"`, `${title} draft`], + ['"release"', '"release draft"'], + ['"release draft"', '"release"'], + ['"release plan"', 'release draft'], + ['release plan', 'release draft'], + ['"release draft plan"', 'release plan'], + ['"release plan draft"', '"release plan"'], + ['"draft release"', 'release'], + ['""', 'draft'], + ['" "', 'plan'], + [`"${title}" or "foreign"`, `${title} draft`], + [`"${title}" and another plan`, `${title} draft`], + [`"${title}`, `${title} draft`], + [`${title}"`, `${title} draft`], + ['"future plan"', 'future plan'], + ['future', 'future plan'], + ['previous', 'previous draft'], + ['"previous draft"', 'previous draft'], + ['"another draft"', 'another draft'], + ['"next plan"', 'next plan'], +]; +for (const [declared, recorded] of negatives) { + test(`target cannot borrow a named or historical match: ${declared} / ${recorded}`, () => { + const f = clone(); f.options.stateEvidence.records[0].question_summary = `Select mode for ${recorded}`; + message(f).text = `Decision: HOLD SCOPE for ${declared}.`; + expect(decide(f)).toBeNull(); + }); + test(`later agreeing Mode does not erase invalid target: ${declared} / ${recorded}`, () => { + const f = clone(); f.options.stateEvidence.records[0].question_summary = `Select mode for ${recorded}`; + message(f).text = `Decision: HOLD SCOPE for ${declared}.\n\nMode: HOLD SCOPE, auto-selected.`; + expect(decide(f)).toBeNull(); + }); +} + +for (const wrap of [ + (s: string) => `"${s}"`, (s: string) => `“${s}”`, (s: string) => `\`${s}\``, + (s: string) => `> ${s}`, (s: string) => ` ${s}`, (s: string) => `\`\`\`text\n${s}\n\`\`\``, + (s: string) => `Example:\n${s}`, (s: string) => `Previous review:\n${s}`, +]) test(`only an asserted field can own a quoted target: ${wrap('Decision')}`, () => { + const f = clone(); message(f).text = wrap(`Decision: HOLD SCOPE for "${title}".`); + expect(decide(f)).toBeNull(); +}); + +for (const value of [ + `HOLD SCOPE for "${title}" if approved`, `HOLD SCOPE for "${title}", pending approval`, + `not HOLD SCOPE for "${title}"`, `HOLD SCOPE for "${title}"; SCOPE EXPANSION`, + `HOLD SCOPE for "${title}" (withdrawn)`, `HOLD SCOPE for "${title}" (I will select it)`, +]) test(`quoted name cannot hide a lifecycle veto: ${value}`, () => { + const f = clone(); message(f).text = `Decision: ${value}.\n\nMode: HOLD SCOPE.`; + expect(decide(f)).toBeNull(); +}); +for (const suffix of [ + '\n\nCorrection: Mode: SCOPE EXPANSION.', + '\n\nCorrection: I withdraw this decision.', + `\n\nDecision: HOLD SCOPE for "foreign target".`, + '\n\nMode pending: HOLD SCOPE.', +]) test(`a later contradiction remains effective: ${suffix}`, () => { + const f = clone(); message(f).text += suffix; expect(decide(f)).toBeNull(); +}); +for (const [name, mutate] of Object.entries({ + 'missing owned log': (f: any) => { f.options.stateEvidence.records = []; }, + 'duplicate owned log': (f: any) => { f.options.stateEvidence.records.push({ ...f.options.stateEvidence.records[0] }); }, + 'foreign audit session': (f: any) => { f.options.stateEvidence.records[0].session_id = 'foreign'; }, + 'different audit choice': (f: any) => { f.options.stateEvidence.records[0].user_choice = 'SCOPE EXPANSION'; }, + 'wrong preference': (f: any) => { f.options.stateEvidence.preference = 'ask'; }, + 'missing preamble ACK': (f: any) => { f.tools = f.tools.filter((e: any) => !(e.kind === 'result' && e.toolUseId === 'toolu_01KbsH6ybJxbNozwbSXywVbb')); }, + 'native question': (f: any) => { f.transcript.calls.push({sessionId:f.options.sessionId}); }, + 'prose question': (f: any) => { f.options.proseQuestionObserved = true; }, + 'decision before log': (f: any) => { message(f).timestamp = new Date(Date.parse(f.options.stateEvidence.records[0].ts) - 1).toISOString(); }, + 'wrong native session': (f: any) => { f.options.sessionId = 'foreign'; }, + 'future log': (f: any) => { f.options.stateEvidence.records[0].ts = new Date(f.options.now + 1).toISOString(); }, +})) test(`actual quoted target retains ${name} boundary`, () => { + const f = clone(); mutate(f); expect(decide(f)).toBeNull(); +}); +for (const preposition of ['for', 'FOR']) test(`a quoted lifecycle word belongs to its title with ${preposition}`, () => { + const f = clone(); f.options.stateEvidence.records[0].question_summary = 'Select mode for Pending notifications draft'; + message(f).text = `Decision: HOLD SCOPE ${preposition} "Pending notifications".`; + expect(decide(f)?.option).toBe('HOLD SCOPE'); +}); +}); + +describe('auto-decision-state', () => { +const capture = capture_auto_decision_state; +const clone = () => structuredClone(capture) as any; +const qid = 'plan-ceo-review-mode'; +function state(f: any) { + const use = f.tools.find((e: any) => e.input?.command?.includes('gstack-question-log')); + // Synthetic file witness, built from the actual literal request. The original + // run did not retain this file, and is still a failed paid attempt. + const record = JSON.parse(/gstack-question-log '(\{[^\n]*\})'/.exec(use.input.command)![1]!); + record.source = 'agent'; + record.ts = f.tools.find((e: any) => e.kind === 'result' && e.toolUseId === use.toolUseId).timestamp; + return { questionId: qid, preference: 'never-ask' as const, records: [record] }; +} +const decide = (f: any) => findNativeAutoDecision(f.transcript, f.tools, f.options); +const mode = (f: any) => f.transcript.assistantMessages.find((m: any) => m.text.startsWith('**Mode:')); + +test('original captured retry cannot prove a masked log succeeded', () => { + expect(decide(clone())).toBeNull(); +}); +test('actual retry declaration plus a completed owned append proves the chosen mode', () => { + const f = clone(); f.options.stateEvidence = state(f); + const result = decide(f); + expect(result?.option).toBe('HOLD SCOPE'); + expect(result?.stateRecord).toEqual(f.options.stateEvidence.records[0]); + expect(result?.questionLogToolUseId).toBeUndefined(); +}); + +for (const [name, mutate] of Object.entries({ + 'foreign record session': (f: any) => { f.options.stateEvidence.records[0].session_id = 'foreign'; }, + 'wrong question': (f: any) => { f.options.stateEvidence.questionId = 'wrong'; }, + 'wrong skill': (f: any) => { f.options.stateEvidence.records[0].skill = 'plan-eng-review'; }, + 'nonautomatic record': (f: any) => { f.options.stateEvidence.records[0].auto_decided = false; }, + 'string flag': (f: any) => { f.options.stateEvidence.records[0].auto_decided = 'true'; }, + 'wrong source': (f: any) => { f.options.stateEvidence.records[0].source = 'hook'; }, + 'different preference': (f: any) => { f.options.stateEvidence.preference = 'always-ask'; }, + 'missing append': (f: any) => { f.options.stateEvidence.records = []; }, + 'duplicate append': (f: any) => { f.options.stateEvidence.records.push({ ...f.options.stateEvidence.records[0] }); }, + 'contradictory recommendation': (f: any) => { f.options.stateEvidence.records[0].recommended = 'SCOPE EXPANSION'; }, + 'empty summary': (f: any) => { f.options.stateEvidence.records[0].question_summary = ''; }, + 'old record': (f: any) => { f.options.stateEvidence.records[0].ts = new Date(f.options.commandStartedAt - 1).toISOString(); }, + 'future record': (f: any) => { f.options.stateEvidence.records[0].ts = new Date(f.options.now + 1).toISOString(); }, + 'record after declaration': (f: any) => { f.options.stateEvidence.records[0].ts = new Date(Date.parse(mode(f).timestamp) + 1).toISOString(); }, + 'invalid timestamp': (f: any) => { f.options.stateEvidence.records[0].ts = 'invalid'; }, + 'actual native question': (f: any) => { f.transcript.calls.push({ sessionId: f.options.sessionId }); }, + 'actual prose question': (f: any) => { f.options.proseQuestionObserved = true; }, + 'failed preamble': (f: any) => { f.tools.find((e: any) => e.kind === 'result' && e.content?.includes('SKILL_START_PROTO')).isError = true; }, + 'quoted declaration': (f: any) => { mode(f).text = '> Mode: HOLD SCOPE (saved preference).'; }, + 'conditional declaration': (f: any) => { mode(f).text = 'Mode: HOLD SCOPE (if approved).'; }, + 'later withdrawal': (f: any) => { mode(f).text += '\n\nCorrection: I withdraw this decision.'; }, + 'later different mode': (f: any) => { mode(f).text += '\n\nMode: SCOPE EXPANSION (saved preference).'; }, +})) test(`owned log witness rejects ${name}`, () => { + const f = clone(); f.options.stateEvidence = state(f); mutate(f); expect(decide(f)).toBeNull(); +}); + +function withState(check: (x: { root: string; project: string; pref: string; log: string; bind: () => ReturnType }) => void) { + const root = fs.realpathSync(fs.mkdtempSync(path.join(os.tmpdir(), 'auto-state-'))); + const project = path.join(root, 'projects', 'fixture'); fs.mkdirSync(project, { recursive: true }); + const pref = path.join(project, 'question-preferences.json'), log = path.join(project, 'question-log.jsonl'); + fs.writeFileSync(pref, JSON.stringify({ [qid]: 'never-ask' })); + const bind = () => bindAutoDecisionState({ stateRoot: root, projectSlug: 'fixture' }, { GSTACK_STATE_ROOT: root }, 'plan-ceo-review'); + try { check({ root, project, pref, log, bind }); } finally { fs.rmSync(root, { recursive: true, force: true }); } +} +test('state witness binds before launch and observes only completed owned file contents', () => withState(({ log, bind }) => { + const read = bind(); expect(read()).toBeUndefined(); + const record = state(clone()).records[0]; fs.writeFileSync(log, JSON.stringify(record) + '\n'); + expect(read()?.records).toEqual([record]); +})); +for (const scenario of ['existing-log', 'preference-change', 'malformed-log', 'log-symlink', 'preference-symlink', 'wrong-root', 'path-escape']) + test(`state binding rejects ${scenario}`, () => withState(({ root, pref, log, bind }) => { + if (scenario === 'wrong-root' || scenario === 'path-escape') { + expect(() => bindAutoDecisionState({ stateRoot: root, projectSlug: scenario === 'path-escape' ? '../fixture' : 'fixture' }, + { GSTACK_STATE_ROOT: scenario === 'wrong-root' ? root + '-other' : root }, 'plan-ceo-review')).toThrow(); return; + } + if (scenario === 'existing-log') { fs.writeFileSync(log, '{}\n'); expect(bind).toThrow('fresh attempt'); return; } + const read = bind(); + if (scenario === 'preference-change') fs.writeFileSync(pref, JSON.stringify({ [qid]: 'always-ask' })); + if (scenario === 'malformed-log') fs.writeFileSync(log, '{'); + if (scenario === 'log-symlink') fs.symlinkSync(pref, log); + if (scenario === 'preference-symlink') { fs.renameSync(pref, pref + '.real'); fs.symlinkSync(pref + '.real', pref); } + expect(read()).toBeUndefined(); + })); +}); diff --git a/test/outside-background-ai.test.ts b/test/outside-background-ai.test.ts deleted file mode 100644 index df0ce706f..000000000 --- a/test/outside-background-ai.test.ts +++ /dev/null @@ -1,103 +0,0 @@ -import { describe, expect, test } from 'bun:test'; -import { claudeOutsideExecutions, foundInvoiceAuthorizationDefect, outsideExecutionTranscript } from './helpers/outside-voice-evidence'; -import captured from './fixtures/outside-background-ai.json'; - -const events = () => structuredClone(captured.events) as any[]; -const block = (input: any[], index: number) => input[index].message.content[0]; -const outputFile = /([^<]+)<\/output-file>/.exec(captured.events[2].attachment!.prompt!)![1]!; -const finding = 'The missing invoice owner authorization check allows another user to access the invoice.'; -const detected = (input: any[]) => foundInvoiceAuthorizationDefect(claudeOutsideExecutions(input), 'codex'); - -function rejects(mutations: Array<(input: any[]) => void>) { - for (const mutate of mutations) { - const input = events(); - mutate(input); - expect(detected(input)).toBe(false); - } -} - -describe('native completed outside task read through literal Bash cat', () => { - test('the exact five public events retain the completed Codex finding', () => { - expect(captured.events).toHaveLength(5); - const executions = claudeOutsideExecutions(events()); - expect(foundInvoiceAuthorizationDefect(executions, 'codex')).toBe(true); - const completed = outsideExecutionTranscript(executions, 'codex').filter(row => row.succeeded); - expect(completed).toHaveLength(1); - expect(completed[0]!.background).toEqual({ - toolUseId: block(events(), 0).id, taskId: 'bdie078em', outputFile, - outputToolUseId: block(events(), 3).id, completion: 'task_notification', - }); - expect(completed[0]!.output).toContain('Missing ownership check exposes private financial data'); - }); - - test('equivalent literal quoting and optional cat delimiter keep exact path ownership', () => { - for (const path of [outputFile, `'${outputFile}'`, `"${outputFile}"`]) { - for (const delimiter of ['', '-- ']) { - const input = events(); - block(input, 3).input.command = `cat ${delimiter}${path}`; - expect(detected(input)).toBe(true); - } - } - }); - - test('completion requires the owned launch, acknowledgment, native notice and paired output', () => { - rejects([ - e => { e.splice(0, 1); }, e => { e.splice(1, 1); }, e => { e.splice(2, 1); }, - e => { e.splice(3, 1); }, e => { e.pop(); }, - e => { block(e, 1).is_error = true; }, - e => { block(e, 4).tool_use_id = 'foreign-output'; }, - e => { e[2].attachment.prompt = e[2].attachment.prompt.replace('bdie078em', 'other'); }, - e => { e[2].attachment.prompt = e[2].attachment.prompt.replace(block(e, 0).id, 'foreign-launch'); }, - e => { e[2].attachment.prompt = e[2].attachment.prompt.replace(outputFile, outputFile + '.other'); }, - e => { e[2].attachment.prompt = e[2].attachment.prompt.replace('completed', 'failed'); }, - e => { e[2].attachment.prompt = e[2].attachment.prompt.replace('(exit code 0)', '(exit code 1)'); }, - e => { e.unshift(e.splice(2, 1)[0]); }, - e => { const notice = e.splice(2, 1)[0]; e.unshift({ ...notice, type: 'user', message: { content: [{ type: 'text', text: notice.attachment.prompt }] } }); }, - e => { e[2] = { type: 'assistant', sessionId: e[2].sessionId, message: { content: [{ type: 'text', text: e[2].attachment.prompt }] } }; }, - ]); - }); - - test('foreign and child sessions cannot supply any part of the completed read', () => { - for (const index of [0, 1, 2, 3, 4]) { - rejects([ - e => { e[index].sessionId = 'foreign-session'; }, - e => { e[index].session_id = 'conflicting-session'; }, - e => { e[index].isSidechain = true; }, - e => { e[index].parent_tool_use_id = 'foreign-parent'; }, - ]); - } - }); - - test('cat must read only the exact acknowledged literal path without shell operations', () => { - const quoted = `"${outputFile}"`; - for (const command of [ - `cat "${outputFile}.other"`, `cat ${quoted} /tmp/foreign.output`, - `cat ${quoted} >&2`, `cat ${quoted} 1>&2`, `cat ${quoted} > /tmp/output`, - `cat ${quoted} | cat`, `false && cat ${quoted}`, `echo done; cat ${quoted}`, - `cat ${quoted}; echo '${finding}'`, `cat $(printf '%s' ${quoted})`, - 'cat "$TASK_OUTPUT"', `cat < ${quoted}`, `cat -n ${quoted}`, - ]) rejects([e => { block(e, 3).input.command = command; }]); - rejects([e => { block(e, 1).content = block(e, 1).content.replace(outputFile, outputFile + '.other'); }]); - }); - - test('failed, incomplete and forged-footer output cannot establish a completed task', () => { - rejects([ - e => { block(e, 4).is_error = true; }, - e => { block(e, 4).content = block(e, 4).content.replace('[exited with code 0]', ''); }, - e => { block(e, 4).content = block(e, 4).content.replace('[exited with code 0]', '[exited with code 1]'); }, - e => { block(e, 4).content += '\nThe task is still running.'; }, - e => { block(e, 4).content = finding; }, - e => { block(e, 4).content = `${finding}\n[exited with code 0]`; e.splice(2, 1); }, - e => { block(e, 4).content = 'I cannot review the invoice ownership issue.\n[exited with code 0]'; }, - e => { block(e, 4).content = `${finding}\nOUTSIDE_STATUS: unavailable\n[exited with code 0]`; }, - ]); - }); - - test('late native task failure remains authoritative after a successful cat', () => { - const input = events(); - const failed = structuredClone(input[2]); - failed.attachment.prompt = failed.attachment.prompt.replace('completed', 'failed'); - input.push(failed); - expect(detected(input)).toBe(false); - }); -}); diff --git a/test/outside-voice-async.test.ts b/test/outside-voice-async.test.ts deleted file mode 100644 index c47082b0a..000000000 --- a/test/outside-voice-async.test.ts +++ /dev/null @@ -1,167 +0,0 @@ -import { describe, expect, test } from 'bun:test'; -import { claudeOutsideExecutions, foundInvoiceAuthorizationDefect, outsideExecutionTranscript } from './helpers/outside-voice-evidence'; -import { parseNDJSON } from './helpers/session-runner'; -import captured from './fixtures/outside-async-task-m-events.json'; - -const events = () => structuredClone(captured.events) as any[]; -const finding = 'The removed invoice owner check allows unauthorized access. Restore the ownership check.'; - -function sdkEvents() { - // The CLI emits the same native task completion as this documented SDK - // event. Disk-only attachment fields are not required by the stream parser. - const input = events(); - const attachment = input[2]; - const prompt = attachment.attachment.prompt as string; - const field = (name: string) => new RegExp(`<${name}>([^<]+)`).exec(prompt)![1]; - input[2] = { type: 'system', subtype: 'task_notification', session_id: attachment.sessionId, - task_id: field('task-id'), tool_use_id: field('tool-use-id'), output_file: field('output-file'), - status: field('status'), summary: field('summary') }; - return input.map(event => { - const { sessionId, isSidechain, ...rest } = event; - return { ...rest, session_id: event.session_id ?? sessionId }; - }); -} - -function detected(input: unknown[]) { - return foundInvoiceAuthorizationDefect(claudeOutsideExecutions(input), 'codex'); -} - -describe('completed background outside execution correlation', () => { - test('the exact task completion and complete output Read establish the actual external finding', () => { - const result = claudeOutsideExecutions(events()); - expect(result).toHaveLength(2); - expect(result[0]!.succeeded).toBe(false); - expect(result[0]!.background?.completion).toBe('pending'); - expect(result[1]!.succeeded).toBe(true); - expect(result[1]!.background).toEqual({ - toolUseId: 'toolu_016PugkAqmFq5DjPFmvDA3DB', taskId: 'bn8jkc52l', - outputFile: events()[3].message.content[0].input.file_path, - outputToolUseId: 'toolu_01J2PNv3YyETiCWXKkHxFVxa', completion: 'task_notification', - }); - expect(result[1]!.output).toContain('Removing the ownership check lets any authenticated caller retrieve any invoice'); - expect(detected(events())).toBe(true); - expect(outsideExecutionTranscript(result, 'codex')[1]!.background).toEqual(result[1]!.background); - }); - - test('SDK NDJSON retains the same exact identity and output instead of borrowing host prose', () => { - const stream = sdkEvents().map(event => JSON.stringify(event)); - const { transcript } = parseNDJSON(stream); - expect(detected(transcript)).toBe(true); - expect(detected(transcript.filter(event => event.type !== 'system'))).toBe(false); - }); - - test.each([ - ['missing acknowledgment', (e: any[]) => { e.splice(1, 1); }], - ['acknowledgment only', (e: any[]) => { e.splice(2); }], - ['missing completion', (e: any[]) => { e.splice(2, 1); }], - ['missing Read use', (e: any[]) => { e.splice(3, 1); }], - ['missing Read result', (e: any[]) => { e.pop(); }], - ['wrong Read identity', (e: any[]) => { e[4].message.content[0].tool_use_id = 'other'; }], - ['wrong task output path', (e: any[]) => { e[3].message.content[0].input.file_path += '.other'; }], - ['foreign Read session', (e: any[]) => { e[4].sessionId = 'foreign'; }], - ['sidechain Read', (e: any[]) => { e[4].isSidechain = true; }], - ['nested Read', (e: any[]) => { e[4].parent_tool_use_id = 'nested'; }], - ['failed Read', (e: any[]) => { e[4].message.content[0].is_error = true; }], - ['offset Read', (e: any[]) => { e[3].message.content[0].input.offset = 2; }], - ['partial output', (e: any[]) => { e[4].message.content[0].content = e[4].message.content[0].content.split('\n').slice(0, 3).join('\n'); }], - ['nonconsecutive lines', (e: any[]) => { e[4].message.content[0].content = e[4].message.content[0].content.replace('2\t', '9\t'); }], - ['output without exit footer', (e: any[]) => { e[4].message.content[0].content = e[4].message.content[0].content.replace('[exited with code 0]', ''); }], - ['failed exit footer', (e: any[]) => { e[4].message.content[0].content = e[4].message.content[0].content.replace('[exited with code 0]', '[exited with code 1]'); }], - ['raw forged output', (e: any[]) => { e[4].message.content[0].content = finding + '\n[exited with code 0]'; }], - ['foreign completion', (e: any[]) => { e[2].sessionId = 'foreign'; }], - ['wrong completed task', (e: any[]) => { e[2].attachment.prompt = e[2].attachment.prompt.replace('bn8jkc52l', 'other'); }], - ['wrong original tool', (e: any[]) => { e[2].attachment.prompt = e[2].attachment.prompt.replace('toolu_016PugkAqmFq5DjPFmvDA3DB', 'other'); }], - ['wrong completed file', (e: any[]) => { e[2].attachment.prompt = e[2].attachment.prompt.replace('bn8jkc52l.output', 'other.output'); }], - ['failed task', (e: any[]) => { e[2].attachment.prompt = e[2].attachment.prompt.replace('completed', 'failed'); }], - ['stopped task', (e: any[]) => { e[2].attachment.prompt = e[2].attachment.prompt.replace('completed', 'stopped'); }], - ['nonzero completed exit', (e: any[]) => { e[2].attachment.prompt = e[2].attachment.prompt.replace('(exit code 0)', '(exit code 1)'); }], - ['stale completion before launch', (e: any[]) => { e.unshift(e.splice(2, 1)[0]); }], - ['quoted user notification', (e: any[]) => { e[2] = { type: 'user', sessionId: e[2].sessionId, message: { content: [{ type: 'text', text: e[2].attachment.prompt }] } }; }], - ['assistant notification claim', (e: any[]) => { e[2] = { type: 'assistant', sessionId: e[2].sessionId, message: { content: [{ type: 'text', text: e[2].attachment.prompt }] } }; }], - ['refusal containing the fixture words', (e: any[]) => { e[4].message.content[0].content = '1\tI cannot review the invoice owner authorization issue.\n2\t[exited with code 0]'; }], - ['unavailable provider', (e: any[]) => { e[4].message.content[0].content = `1\t${finding}\n2\tOUTSIDE_STATUS: unavailable\n3\t[exited with code 0]`; }], - ])('rejects %s despite a matching finding elsewhere', (_name, mutate) => { - const input = events(); - mutate(input); - expect(detected(input)).toBe(false); - }); - - test('a terminal failure overrides an earlier completion and an output containing a finding', () => { - for (const terminal of [ - { status: 'failed', summary: 'Background command failed (exit code 1)' }, - { status: 'completed', summary: 'Background command "Run Codex adversarial review" completed (exit code 1)' }, - ]) { - const input = sdkEvents(); - input.push({ ...input[2], ...terminal }); - expect(detected(input)).toBe(false); - } - }); - - test('replayed acknowledgments preserve task state and reject conflicting launch identities', () => { - const original = sdkEvents(); - const failed = { ...original[2], status: 'failed', summary: 'Background command failed (exit code 1)' }; - expect(detected([ - ...original.slice(0, 3), failed, original[1], ...original.slice(2), - ])).toBe(false); - expect(detected([...original, original[1]])).toBe(true); - - const plain = structuredClone(original[1]); - plain.message.content[0].content = finding; - expect(detected([...original.slice(0, 2), plain])).toBe(false); - const retained = claudeOutsideExecutions([...original, plain]); - expect(retained.filter(result => result.succeeded)).toHaveLength(1); - expect(retained.find(result => result.output === finding)?.succeeded).toBe(false); - plain.message.content[0].is_error = true; - plain.message.content[0].content = 'The background launch failed.'; - expect(detected([...original, plain])).toBe(false); - - for (const alter of [ - (ack: any) => { ack.message.content[0].content = ack.message.content[0].content.replaceAll('bn8jkc52l', 'other-task'); }, - (ack: any) => { ack.session_id = 'foreign'; }, - (ack: any) => { ack.message.content[0].is_error = true; }, - ]) { - const acknowledgment = structuredClone(original[1]); - alter(acknowledgment); - expect(detected([...original, acknowledgment])).toBe(false); - } - }); - - test('partial and failed exact-path Reads remain diagnostic output without earning completion', () => { - for (const fail of [false, true]) { - const input = events(); - input[4].message.content[0].content = finding; - input[4].message.content[0].is_error = fail; - const retained = outsideExecutionTranscript(claudeOutsideExecutions(input), 'codex'); - expect(retained.at(-1)!.output).toBe(finding); - expect(retained.at(-1)!.succeeded).toBe(false); - expect(detected(input)).toBe(false); - } - }); - - test('an unsupported background acknowledgment cannot pass as a foreground result', () => { - const input = events().slice(0, 2); - input[1].message.content[0].content += '\n' + finding; - expect(detected(input)).toBe(false); - }); - - test('TaskOutput requires the exact launched task and native completed exit-zero envelope', () => { - const input = sdkEvents().slice(0, 2); - const session_id = input[0].session_id; - const result = `success\n\nbn8jkc52l\n\nlocal_bash\n\ncompleted\n\n0\n\n\n${finding}\n`; - input.push({ type: 'assistant', session_id, message: { content: [{ type: 'tool_use', name: 'TaskOutput', id: 'task-output', input: { task_id: 'bn8jkc52l' } }] } }); - input.push({ type: 'user', session_id, message: { content: [{ type: 'tool_result', tool_use_id: 'task-output', content: result }] } }); - expect(detected(input)).toBe(true); - for (const content of [ - result.replace('>success<', '>not_ready<'), result.replace('>completed<', '>running<'), - result.replace('>0<', '>1<'), result.replace('>bn8jkc52l<', '>other<'), - result.replace('>local_bash<', '>local_agent<'), finding + result, - ]) { - const invalid = structuredClone(input); - invalid[3].message.content[0].content = content; - expect(detected(invalid)).toBe(false); - } - const invalid = structuredClone(input); - invalid[3].message.content[0].is_error = true; - expect(detected(invalid)).toBe(false); - }); -}); diff --git a/test/outside-voice-evidence.test.ts b/test/outside-voice-evidence.test.ts index 0f3eb128c..2d0f6249c 100644 --- a/test/outside-voice-evidence.test.ts +++ b/test/outside-voice-evidence.test.ts @@ -4,6 +4,9 @@ import * as os from 'node:os'; import * as path from 'node:path'; import { EvalCollector } from './helpers/eval-store'; import { claudeOutsideExecutions, codexOutsideExecutions, foundInvoiceAuthorizationDefect, outsideExecutionTranscript } from './helpers/outside-voice-evidence'; +import captured_outside_background_ai from './fixtures/outside-background-ai.json'; +import { parseNDJSON } from './helpers/session-runner'; +import captured_outside_voice_async from './fixtures/outside-async-task-m-events.json'; const finding = '[P1] invoice.ts removed the owner check, allowing unauthorized access to private invoices.'; @@ -76,3 +79,272 @@ describe('outside execution artifact retention', () => { ]), 'codex')).toEqual([]); }); }); + +describe('outside-background-ai', () => { +const captured = captured_outside_background_ai; +const events = () => structuredClone(captured.events) as any[]; +const block = (input: any[], index: number) => input[index].message.content[0]; +const outputFile = /([^<]+)<\/output-file>/.exec(captured.events[2].attachment!.prompt!)![1]!; +const finding = 'The missing invoice owner authorization check allows another user to access the invoice.'; +const detected = (input: any[]) => foundInvoiceAuthorizationDefect(claudeOutsideExecutions(input), 'codex'); + +function rejects(mutations: Array<(input: any[]) => void>) { + for (const mutate of mutations) { + const input = events(); + mutate(input); + expect(detected(input)).toBe(false); + } +} + +describe('native completed outside task read through literal Bash cat', () => { + test('the exact five public events retain the completed Codex finding', () => { + expect(captured.events).toHaveLength(5); + const executions = claudeOutsideExecutions(events()); + expect(foundInvoiceAuthorizationDefect(executions, 'codex')).toBe(true); + const completed = outsideExecutionTranscript(executions, 'codex').filter(row => row.succeeded); + expect(completed).toHaveLength(1); + expect(completed[0]!.background).toEqual({ + toolUseId: block(events(), 0).id, taskId: 'bdie078em', outputFile, + outputToolUseId: block(events(), 3).id, completion: 'task_notification', + }); + expect(completed[0]!.output).toContain('Missing ownership check exposes private financial data'); + }); + + test('equivalent literal quoting and optional cat delimiter keep exact path ownership', () => { + for (const path of [outputFile, `'${outputFile}'`, `"${outputFile}"`]) { + for (const delimiter of ['', '-- ']) { + const input = events(); + block(input, 3).input.command = `cat ${delimiter}${path}`; + expect(detected(input)).toBe(true); + } + } + }); + + test('completion requires the owned launch, acknowledgment, native notice and paired output', () => { + rejects([ + e => { e.splice(0, 1); }, e => { e.splice(1, 1); }, e => { e.splice(2, 1); }, + e => { e.splice(3, 1); }, e => { e.pop(); }, + e => { block(e, 1).is_error = true; }, + e => { block(e, 4).tool_use_id = 'foreign-output'; }, + e => { e[2].attachment.prompt = e[2].attachment.prompt.replace('bdie078em', 'other'); }, + e => { e[2].attachment.prompt = e[2].attachment.prompt.replace(block(e, 0).id, 'foreign-launch'); }, + e => { e[2].attachment.prompt = e[2].attachment.prompt.replace(outputFile, outputFile + '.other'); }, + e => { e[2].attachment.prompt = e[2].attachment.prompt.replace('completed', 'failed'); }, + e => { e[2].attachment.prompt = e[2].attachment.prompt.replace('(exit code 0)', '(exit code 1)'); }, + e => { e.unshift(e.splice(2, 1)[0]); }, + e => { const notice = e.splice(2, 1)[0]; e.unshift({ ...notice, type: 'user', message: { content: [{ type: 'text', text: notice.attachment.prompt }] } }); }, + e => { e[2] = { type: 'assistant', sessionId: e[2].sessionId, message: { content: [{ type: 'text', text: e[2].attachment.prompt }] } }; }, + ]); + }); + + test('foreign and child sessions cannot supply any part of the completed read', () => { + for (const index of [0, 1, 2, 3, 4]) { + rejects([ + e => { e[index].sessionId = 'foreign-session'; }, + e => { e[index].session_id = 'conflicting-session'; }, + e => { e[index].isSidechain = true; }, + e => { e[index].parent_tool_use_id = 'foreign-parent'; }, + ]); + } + }); + + test('cat must read only the exact acknowledged literal path without shell operations', () => { + const quoted = `"${outputFile}"`; + for (const command of [ + `cat "${outputFile}.other"`, `cat ${quoted} /tmp/foreign.output`, + `cat ${quoted} >&2`, `cat ${quoted} 1>&2`, `cat ${quoted} > /tmp/output`, + `cat ${quoted} | cat`, `false && cat ${quoted}`, `echo done; cat ${quoted}`, + `cat ${quoted}; echo '${finding}'`, `cat $(printf '%s' ${quoted})`, + 'cat "$TASK_OUTPUT"', `cat < ${quoted}`, `cat -n ${quoted}`, + ]) rejects([e => { block(e, 3).input.command = command; }]); + rejects([e => { block(e, 1).content = block(e, 1).content.replace(outputFile, outputFile + '.other'); }]); + }); + + test('failed, incomplete and forged-footer output cannot establish a completed task', () => { + rejects([ + e => { block(e, 4).is_error = true; }, + e => { block(e, 4).content = block(e, 4).content.replace('[exited with code 0]', ''); }, + e => { block(e, 4).content = block(e, 4).content.replace('[exited with code 0]', '[exited with code 1]'); }, + e => { block(e, 4).content += '\nThe task is still running.'; }, + e => { block(e, 4).content = finding; }, + e => { block(e, 4).content = `${finding}\n[exited with code 0]`; e.splice(2, 1); }, + e => { block(e, 4).content = 'I cannot review the invoice ownership issue.\n[exited with code 0]'; }, + e => { block(e, 4).content = `${finding}\nOUTSIDE_STATUS: unavailable\n[exited with code 0]`; }, + ]); + }); + + test('late native task failure remains authoritative after a successful cat', () => { + const input = events(); + const failed = structuredClone(input[2]); + failed.attachment.prompt = failed.attachment.prompt.replace('completed', 'failed'); + input.push(failed); + expect(detected(input)).toBe(false); + }); +}); +}); + +describe('outside-voice-async', () => { +const captured = captured_outside_voice_async; +const events = () => structuredClone(captured.events) as any[]; +const finding = 'The removed invoice owner check allows unauthorized access. Restore the ownership check.'; + +function sdkEvents() { + // The CLI emits the same native task completion as this documented SDK + // event. Disk-only attachment fields are not required by the stream parser. + const input = events(); + const attachment = input[2]; + const prompt = attachment.attachment.prompt as string; + const field = (name: string) => new RegExp(`<${name}>([^<]+)`).exec(prompt)![1]; + input[2] = { type: 'system', subtype: 'task_notification', session_id: attachment.sessionId, + task_id: field('task-id'), tool_use_id: field('tool-use-id'), output_file: field('output-file'), + status: field('status'), summary: field('summary') }; + return input.map(event => { + const { sessionId, isSidechain, ...rest } = event; + return { ...rest, session_id: event.session_id ?? sessionId }; + }); +} + +function detected(input: unknown[]) { + return foundInvoiceAuthorizationDefect(claudeOutsideExecutions(input), 'codex'); +} + +describe('completed background outside execution correlation', () => { + test('the exact task completion and complete output Read establish the actual external finding', () => { + const result = claudeOutsideExecutions(events()); + expect(result).toHaveLength(2); + expect(result[0]!.succeeded).toBe(false); + expect(result[0]!.background?.completion).toBe('pending'); + expect(result[1]!.succeeded).toBe(true); + expect(result[1]!.background).toEqual({ + toolUseId: 'toolu_016PugkAqmFq5DjPFmvDA3DB', taskId: 'bn8jkc52l', + outputFile: events()[3].message.content[0].input.file_path, + outputToolUseId: 'toolu_01J2PNv3YyETiCWXKkHxFVxa', completion: 'task_notification', + }); + expect(result[1]!.output).toContain('Removing the ownership check lets any authenticated caller retrieve any invoice'); + expect(detected(events())).toBe(true); + expect(outsideExecutionTranscript(result, 'codex')[1]!.background).toEqual(result[1]!.background); + }); + + test('SDK NDJSON retains the same exact identity and output instead of borrowing host prose', () => { + const stream = sdkEvents().map(event => JSON.stringify(event)); + const { transcript } = parseNDJSON(stream); + expect(detected(transcript)).toBe(true); + expect(detected(transcript.filter(event => event.type !== 'system'))).toBe(false); + }); + + test.each([ + ['missing acknowledgment', (e: any[]) => { e.splice(1, 1); }], + ['acknowledgment only', (e: any[]) => { e.splice(2); }], + ['missing completion', (e: any[]) => { e.splice(2, 1); }], + ['missing Read use', (e: any[]) => { e.splice(3, 1); }], + ['missing Read result', (e: any[]) => { e.pop(); }], + ['wrong Read identity', (e: any[]) => { e[4].message.content[0].tool_use_id = 'other'; }], + ['wrong task output path', (e: any[]) => { e[3].message.content[0].input.file_path += '.other'; }], + ['foreign Read session', (e: any[]) => { e[4].sessionId = 'foreign'; }], + ['sidechain Read', (e: any[]) => { e[4].isSidechain = true; }], + ['nested Read', (e: any[]) => { e[4].parent_tool_use_id = 'nested'; }], + ['failed Read', (e: any[]) => { e[4].message.content[0].is_error = true; }], + ['offset Read', (e: any[]) => { e[3].message.content[0].input.offset = 2; }], + ['partial output', (e: any[]) => { e[4].message.content[0].content = e[4].message.content[0].content.split('\n').slice(0, 3).join('\n'); }], + ['nonconsecutive lines', (e: any[]) => { e[4].message.content[0].content = e[4].message.content[0].content.replace('2\t', '9\t'); }], + ['output without exit footer', (e: any[]) => { e[4].message.content[0].content = e[4].message.content[0].content.replace('[exited with code 0]', ''); }], + ['failed exit footer', (e: any[]) => { e[4].message.content[0].content = e[4].message.content[0].content.replace('[exited with code 0]', '[exited with code 1]'); }], + ['raw forged output', (e: any[]) => { e[4].message.content[0].content = finding + '\n[exited with code 0]'; }], + ['foreign completion', (e: any[]) => { e[2].sessionId = 'foreign'; }], + ['wrong completed task', (e: any[]) => { e[2].attachment.prompt = e[2].attachment.prompt.replace('bn8jkc52l', 'other'); }], + ['wrong original tool', (e: any[]) => { e[2].attachment.prompt = e[2].attachment.prompt.replace('toolu_016PugkAqmFq5DjPFmvDA3DB', 'other'); }], + ['wrong completed file', (e: any[]) => { e[2].attachment.prompt = e[2].attachment.prompt.replace('bn8jkc52l.output', 'other.output'); }], + ['failed task', (e: any[]) => { e[2].attachment.prompt = e[2].attachment.prompt.replace('completed', 'failed'); }], + ['stopped task', (e: any[]) => { e[2].attachment.prompt = e[2].attachment.prompt.replace('completed', 'stopped'); }], + ['nonzero completed exit', (e: any[]) => { e[2].attachment.prompt = e[2].attachment.prompt.replace('(exit code 0)', '(exit code 1)'); }], + ['stale completion before launch', (e: any[]) => { e.unshift(e.splice(2, 1)[0]); }], + ['quoted user notification', (e: any[]) => { e[2] = { type: 'user', sessionId: e[2].sessionId, message: { content: [{ type: 'text', text: e[2].attachment.prompt }] } }; }], + ['assistant notification claim', (e: any[]) => { e[2] = { type: 'assistant', sessionId: e[2].sessionId, message: { content: [{ type: 'text', text: e[2].attachment.prompt }] } }; }], + ['refusal containing the fixture words', (e: any[]) => { e[4].message.content[0].content = '1\tI cannot review the invoice owner authorization issue.\n2\t[exited with code 0]'; }], + ['unavailable provider', (e: any[]) => { e[4].message.content[0].content = `1\t${finding}\n2\tOUTSIDE_STATUS: unavailable\n3\t[exited with code 0]`; }], + ])('rejects %s despite a matching finding elsewhere', (_name, mutate) => { + const input = events(); + mutate(input); + expect(detected(input)).toBe(false); + }); + + test('a terminal failure overrides an earlier completion and an output containing a finding', () => { + for (const terminal of [ + { status: 'failed', summary: 'Background command failed (exit code 1)' }, + { status: 'completed', summary: 'Background command "Run Codex adversarial review" completed (exit code 1)' }, + ]) { + const input = sdkEvents(); + input.push({ ...input[2], ...terminal }); + expect(detected(input)).toBe(false); + } + }); + + test('replayed acknowledgments preserve task state and reject conflicting launch identities', () => { + const original = sdkEvents(); + const failed = { ...original[2], status: 'failed', summary: 'Background command failed (exit code 1)' }; + expect(detected([ + ...original.slice(0, 3), failed, original[1], ...original.slice(2), + ])).toBe(false); + expect(detected([...original, original[1]])).toBe(true); + + const plain = structuredClone(original[1]); + plain.message.content[0].content = finding; + expect(detected([...original.slice(0, 2), plain])).toBe(false); + const retained = claudeOutsideExecutions([...original, plain]); + expect(retained.filter(result => result.succeeded)).toHaveLength(1); + expect(retained.find(result => result.output === finding)?.succeeded).toBe(false); + plain.message.content[0].is_error = true; + plain.message.content[0].content = 'The background launch failed.'; + expect(detected([...original, plain])).toBe(false); + + for (const alter of [ + (ack: any) => { ack.message.content[0].content = ack.message.content[0].content.replaceAll('bn8jkc52l', 'other-task'); }, + (ack: any) => { ack.session_id = 'foreign'; }, + (ack: any) => { ack.message.content[0].is_error = true; }, + ]) { + const acknowledgment = structuredClone(original[1]); + alter(acknowledgment); + expect(detected([...original, acknowledgment])).toBe(false); + } + }); + + test('partial and failed exact-path Reads remain diagnostic output without earning completion', () => { + for (const fail of [false, true]) { + const input = events(); + input[4].message.content[0].content = finding; + input[4].message.content[0].is_error = fail; + const retained = outsideExecutionTranscript(claudeOutsideExecutions(input), 'codex'); + expect(retained.at(-1)!.output).toBe(finding); + expect(retained.at(-1)!.succeeded).toBe(false); + expect(detected(input)).toBe(false); + } + }); + + test('an unsupported background acknowledgment cannot pass as a foreground result', () => { + const input = events().slice(0, 2); + input[1].message.content[0].content += '\n' + finding; + expect(detected(input)).toBe(false); + }); + + test('TaskOutput requires the exact launched task and native completed exit-zero envelope', () => { + const input = sdkEvents().slice(0, 2); + const session_id = input[0].session_id; + const result = `success\n\nbn8jkc52l\n\nlocal_bash\n\ncompleted\n\n0\n\n\n${finding}\n`; + input.push({ type: 'assistant', session_id, message: { content: [{ type: 'tool_use', name: 'TaskOutput', id: 'task-output', input: { task_id: 'bn8jkc52l' } }] } }); + input.push({ type: 'user', session_id, message: { content: [{ type: 'tool_result', tool_use_id: 'task-output', content: result }] } }); + expect(detected(input)).toBe(true); + for (const content of [ + result.replace('>success<', '>not_ready<'), result.replace('>completed<', '>running<'), + result.replace('>0<', '>1<'), result.replace('>bn8jkc52l<', '>other<'), + result.replace('>local_bash<', '>local_agent<'), finding + result, + ]) { + const invalid = structuredClone(input); + invalid[3].message.content[0].content = content; + expect(detected(invalid)).toBe(false); + } + const invalid = structuredClone(input); + invalid[3].message.content[0].is_error = true; + expect(detected(invalid)).toBe(false); + }); +}); +}); diff --git a/test/paid-free-boundary.test.ts b/test/paid-free-boundary.test.ts index 5a26d43c6..71a5350d3 100644 --- a/test/paid-free-boundary.test.ts +++ b/test/paid-free-boundary.test.ts @@ -147,7 +147,7 @@ describe('paid/free dependency boundary', () => { 'scripts/test-free-shards.ts', 'scripts/test-strict-output.ts', 'test/eng-scope-entry-ap.test.ts', 'test/helpers/plan-floor-review.ts', 'test/plan-floor-permission.test.ts', 'test/plan-floor-review.test.ts', - 'test/plan-review-cases.test.ts', 'test/plan-scope-recovery-av.test.ts', + 'test/plan-review-cases.test.ts', 'test/plan-scope-selection.test.ts', 'test/strict-output-formats.test.ts', ]; const result = computePaidCaseSelection({ profile: 'pr', env: {}, changedFiles }); diff --git a/test/paid-overlay-scheduling.test.ts b/test/paid-overlay-scheduling.test.ts index cb9299add..ca683b857 100644 --- a/test/paid-overlay-scheduling.test.ts +++ b/test/paid-overlay-scheduling.test.ts @@ -62,7 +62,7 @@ describe('overlay file policy', () => { expect(buildPaidShardArgs([file], resolvePaidShardTimeoutMs([file]), 2, retriesForFiles([file]))) .toContain('--timeout=1830000'); } - for (const file of [normalFile, 'test/skill-e2e-overlay-harness.test.ts', 'test/model-overlay-opus-4-7.test.ts']) { + for (const file of [normalFile, 'test/skill-e2e-overlay-harness.test.ts', 'test/model-overlays.test.ts']) { expect(isOverlayTestFile(file)).toBe(false); expect(resolvePaidShardTimeoutMs([file])).toBe(DEFAULT_SHARD_TIMEOUT_MS); expect(retriesForFiles([file])).toBe(1); diff --git a/test/periodic-fixture-selection.test.ts b/test/periodic-fixture-selection.test.ts index 2de88e790..e35c395c2 100644 --- a/test/periodic-fixture-selection.test.ts +++ b/test/periodic-fixture-selection.test.ts @@ -310,8 +310,8 @@ test('same-plan expansion disposition replay selects the existing mode helper co test('structured auto-decision evidence selects every native observer', () => { const expected = ['auto-decide-preserved', 'plan-ceo-review-plan-mode', 'plan-design-review-plan-mode', 'plan-devex-review-plan-mode', 'plan-eng-review-plan-mode', 'plan-mode-no-op']; - for (const file of ['test/auto-decide-structured.test.ts', 'test/fixtures/auto-decide-structured-77.json', - 'test/helpers/auto-decision-state.ts', 'test/auto-decision-state.test.ts', 'test/fixtures/auto-decide-state-cab3.json']) { + for (const file of ['test/fixtures/auto-decide-structured-77.json', + 'test/helpers/auto-decision-state.ts', 'test/fixtures/auto-decide-state-cab3.json']) { expect([...selectTests([file], E2E_TOUCHFILES).selected].sort()).toEqual(expected); expect(selectTests([file], LLM_JUDGE_TOUCHFILES).selected).toEqual([]); } @@ -326,8 +326,8 @@ test('explanatory native mode evidence selects all observers with their existing const expected = ['auto-decide-preserved', 'office-hours-auto-mode', 'plan-ceo-review-plan-mode', 'plan-design-review-plan-mode', 'plan-devex-review-plan-mode', 'plan-eng-review-plan-mode', 'plan-mode-no-op']; - for (const file of ['test/helpers/native-auto-decide.ts', 'test/auto-decide-current-declaration.test.ts', - 'test/fixtures/auto-decide-current-declaration-6aef.json', 'test/auto-decide-explanatory-mode.test.ts', + for (const file of ['test/helpers/native-auto-decide.ts', 'test/native-auto-decide.test.ts', + 'test/fixtures/auto-decide-current-declaration-6aef.json', 'test/native-auto-decide.test.ts', 'test/fixtures/auto-decide-explanatory-mode-043a.json', 'test/fixtures/auto-decide-explanatory-mode-749df.json']) { expect([...selectTests([file], E2E_TOUCHFILES).selected].sort()).toEqual(expected); expect(selectTests([file], LLM_JUDGE_TOUCHFILES).selected).toEqual([]); @@ -409,9 +409,9 @@ const nativeRepairDependencies = [ { "name": "AUTO mode declarations", "files": [ - "test/auto-decide-recommendation-scope.test.ts", + "test/native-auto-decide.test.ts", "test/fixtures/auto-decide-recommendation-361c.json", - "test/auto-decide-target-identity.test.ts", + "test/native-auto-decide.test.ts", "test/fixtures/auto-decide-target-361c.json" ], "owners": [ diff --git a/test/plan-count-completion.test.ts b/test/plan-count-completion.test.ts index cb3a8e6c4..a7bde9fe9 100644 --- a/test/plan-count-completion.test.ts +++ b/test/plan-count-completion.test.ts @@ -9,6 +9,26 @@ import type { PlanCountTranscript } from './helpers/plan-count-transcript'; import capturedL from './fixtures/devex-review-l-calls.json'; import designStatusCapture from './fixtures/design-count-native-issue-fields.json'; import designEnvelope from './fixtures/design-completion-envelope-90f.json'; +import { nativePlanCallFingerprint } from './helpers/claude-pty-runner'; +import type { NativePlanQuestionCall } from './helpers/plan-count-transcript'; +import captured_ceo_completion_handoff_m from './fixtures/ceo-completion-handoff-m-call.json'; +import nextStepCapture_ceo_completion_handoff_m from './fixtures/ceo-handoff-n-calls.json'; +import captured_ceo_completion_handoff_o from './fixtures/ceo-completion-handoff-o-call.json'; +import capturedQ_ceo_completion_handoff_o from './fixtures/ceo-completion-handoff-q-call.json'; +import fs_ceo_handoff_y from 'node:fs'; +import os_ceo_handoff_y from 'node:os'; +import path_ceo_handoff_y from 'node:path'; +import fixture_ceo_handoff_y from './fixtures/ceo-handoff-y-call.json'; +import captured_dx_manual_handoff_ao from './fixtures/dx-manual-handoff-ao.json'; +import { E2E_TOUCHFILES } from './helpers/touchfiles-data'; +import { LLM_JUDGE_TOUCHFILES } from './helpers/touchfiles-data'; +import { GLOBAL_TOUCHFILES } from './helpers/touchfiles-data'; +import fixture_plan_count_dx_handoff_o from './fixtures/devex-handoff-o-call.json'; +import actual_eng_next_handoff_ah from './fixtures/eng-next-handoff-ah.json'; +import { isCurrentPlanApprovalScreen } from './helpers/plan-count-pending-exit'; +import { createHash } from 'node:crypto'; +import capture_eng_task_pause_navigation_f359 from './fixtures/eng-task-pause-navigation-f359.json'; +import fixture_design_count_native_8525 from './fixtures/design-count-native-8525.json'; describe('captured Design completion envelope', () => { function completedEnvelope() { @@ -1011,3 +1031,510 @@ describe('untagged completed DX handoff', () => { } finally {f.cleanup();} }); }); + +describe('ceo-completion-handoff-m', () => { +const captured = captured_ceo_completion_handoff_m; +const nextStepCapture = nextStepCapture_ceo_completion_handoff_m; +const calls = () => structuredClone(captured.calls) as NativePlanQuestionCall[]; +const handoff = () => calls().at(-1)!; +const fingerprint = (call: NativePlanQuestionCall) => nativePlanCallFingerprint(call, 0, false); +describe('CEO completion described by a native navigation choice', () => { +}); + +describe('native next-review navigation with a resolved CEO recap', () => { + const retryCalls = () => structuredClone(captured.distinctRetry.calls) as NativePlanQuestionCall[]; + const retryHandoff = () => retryCalls().at(-1)!; +}); + +describe('CEO completed next-step identity in native option order', () => { + const input = () => structuredClone(nextStepCapture.calls) as NativePlanQuestionCall[]; + const actual = () => input().at(-1)!; + test('the actual report precedes handoff but the captured absent Exit remains incomplete', () => { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'ceo-native-next-step-')); + const report = path.join(dir, 'plan.md'); + try { + fs.writeFileSync(report, nextStepCapture.report.content); + const written = Date.parse(nextStepCapture.report.successfulUpdateAt) / 1000; + fs.utimesSync(report, written, written); + const calls = input(); + expect(Date.parse(calls.at(-2)!.answeredAt!)).toBeLessThan(written * 1000); + expect(Date.parse(calls.at(-1)!.answeredAt!)).toBeGreaterThan(written * 1000); + const transcript = { status: 'ready' as const, calls, assistantMessages: [], + planReadyRequests: structuredClone(nextStepCapture.planReadyRequests) }; + const admin = new Set([fingerprint(calls.at(-1)!).signature]); + expect(hasNativePlanTerminal(transcript, report, Date.parse('2026-09-09T01:06:22Z'), 'plan_ready', admin)).toBe(false); + } finally { + fs.rmSync(dir, { recursive: true, force: true }); + } + }); +}); +}); + +describe('ceo-completion-handoff-o', () => { +const captured = captured_ceo_completion_handoff_o; +const capturedQ = capturedQ_ceo_completion_handoff_o; +const calls = () => structuredClone(captured.calls) as NativePlanQuestionCall[]; +const handoff = () => calls().at(-1)!; +describe('closed CEO navigation with the native review-prefixed identity', () => { +}); + +describe('CEO completion recap after native project metadata', () => { + const qCalls = () => structuredClone(capturedQ.calls) as NativePlanQuestionCall[]; + const qHandoff = () => qCalls().at(-1)!; + test('actual full report and Exit chronology retain last substantive-answer freshness', () => { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'ceo-metadata-navigation-')); + const report = path.join(dir, 'plan.md'); + try { + fs.writeFileSync(report, capturedQ.reportContent); + const reportAt = Date.parse(capturedQ.reportAt) / 1000; + fs.utimesSync(report, reportAt, reportAt); + const transcript = { status: 'ready' as const, calls: qCalls(), assistantMessages: [], planReadyRequests: structuredClone(capturedQ.planReadyRequests) }; + const administrative = new Set([`${qHandoff().sessionId}:${qHandoff().toolUseId}`]); + const start = Date.parse('2026-09-09T03:25:54Z'); + expect(hasNativePlanTerminal(transcript, report, start, 'plan_ready')).toBe(false); + expect(hasNativePlanTerminal(transcript, report, start, 'plan_ready', administrative)).toBe(true); + transcript.planReadyRequests[0]!.failed = true; + expect(hasNativePlanTerminal(transcript, report, start, 'plan_ready', administrative)).toBe(false); + transcript.planReadyRequests = structuredClone(capturedQ.planReadyRequests); + fs.utimesSync(report, start / 1000, start / 1000); + expect(hasNativePlanTerminal(transcript, report, start, 'plan_ready', administrative)).toBe(false); + fs.writeFileSync(report, 'Incomplete plan'); + fs.utimesSync(report, reportAt, reportAt); + expect(hasNativePlanTerminal(transcript, report, start, 'plan_ready', administrative)).toBe(false); + } finally { fs.rmSync(dir, { recursive: true, force: true }); } + }); +}); +}); + +describe('ceo-handoff-y', () => { +const fs = fs_ceo_handoff_y; +const os = os_ceo_handoff_y; +const path = path_ceo_handoff_y; +const fixture = fixture_ceo_handoff_y; +const fp=(c:NativePlanQuestionCall)=>nativePlanCallFingerprint(c,0,false); +describe('Y bare next-Eng navigation is administrative, not completion evidence',()=>{ + test('independent fresh report and native Exit still gate completion; menu alone cannot pass',()=>{ + const dir=fs.mkdtempSync(path.join(os.tmpdir(),'gstack-handoff-y-free-'));const report=path.join(dir,'report.md'); + try{fs.writeFileSync(report,fixture.report);const calls=structuredClone(fixture.calls) as NativePlanQuestionCall[];const transcript={status:'ready' as const,calls,assistantMessages:[],planReadyRequests:structuredClone(fixture.planReadyRequests)};const handoff=calls.at(-1)!;const admin=new Set([fp(handoff).signature]);const issueAt=Date.parse(calls.at(-2)!.answeredAt!),handoffAt=Date.parse(handoff.answeredAt!);const started=Date.parse(calls[0]!.answeredAt!)-1000; + // Controlled metadata only: original Y report mtime was not captured. + const between=(issueAt+handoffAt)/2;fs.utimesSync(report,between/1000,between/1000); + expect(hasNativePlanTerminal(transcript,report,started,'plan_ready')).toBe(false);expect(hasNativePlanTerminal(transcript,report,started,'plan_ready',admin)).toBe(true); + fs.utimesSync(report,(issueAt-1)/1000,(issueAt-1)/1000);expect(hasNativePlanTerminal(transcript,report,started,'plan_ready',admin)).toBe(false); + fs.utimesSync(report,between/1000,between/1000);expect(hasNativePlanTerminal({...transcript,planReadyRequests:[]},report,started,'plan_ready',admin)).toBe(false); + expect(hasNativePlanTerminal({...transcript,calls:[handoff]},report,started,'plan_ready',admin)).toBe(false); + }finally{fs.rmSync(dir,{recursive:true,force:true});} + }); +}); +}); + +describe('dx-manual-handoff-ao', () => { +const fs = fs_ceo_handoff_y; +const os = os_ceo_handoff_y; +const path = path_ceo_handoff_y; +const captured = captured_dx_manual_handoff_ao; +type Edit=(calls:NativePlanQuestionCall[], transcript:PlanCountTranscript, report:string)=>void; +function replay(edit?:Edit){ + const dir=fs.mkdtempSync(path.join(os.tmpdir(),'dx-manual-handoff-ao-')); + try { + const report=path.join(dir,'report.md');fs.writeFileSync(report,captured.reportContent); + const written=captured.provenance.reportMtimeMs/1000;fs.utimesSync(report,written,written); + const transcript={status:'ready',calls:structuredClone(captured.calls),assistantMessages:[],planReadyRequests:structuredClone(captured.planReadyRequests)} as PlanCountTranscript; + edit?.(transcript.calls,transcript,report); + return hasNativePlanTerminal(transcript,report,captured.provenance.startedAt,'plan_ready'); + } finally {fs.rmSync(dir,{recursive:true,force:true});} +} +function change(call:NativePlanQuestionCall,from:string,to:string){ + const q=call.questions[0]!;expect(q.question).toContain(from); + const selected=call.answers![q.question];q.question=q.question.replace(from,to);call.answers={[q.question]:selected!}; +} +describe('AO completed manual DX handoff preserves report freshness',()=>{ + test('shared completion callers register the regression with dense literal paths',()=>{ + const arrays=[...Object.values(E2E_TOUCHFILES),...Object.values(LLM_JUDGE_TOUCHFILES),GLOBAL_TOUCHFILES]; + expect(arrays).toHaveLength(216); + for(const values of arrays)for(let i=0;i{ + expect(captured.calls).toHaveLength(2);expect(captured.events).toHaveLength(4); + expect(Date.parse(captured.calls[0]!.answeredAt!)).toBeLessThan(captured.provenance.reportMtimeMs); + expect(Date.parse(captured.calls[1]!.answeredAt!)).toBeGreaterThan(captured.provenance.reportMtimeMs); + expect(classifyPlanCountFrame(captured.screen)).toBe('plan_ready'); + expect(replay()).toBe(true); + }); + test('equivalent completed recap and manual roles retain current authority',()=>{ + for(const [from,to] of [ + [' (5/10 -> 8.5/10)',''], + ['5/10 -> 8.5/10','6/10 → 9/10'], + ['What should happen next?',"What's next?"], + ['The DX review found','The DX review identified'], + ['All are written into the plan as tasks T1 to T9.','All DX decisions and tasks are recorded in the plan.'], + ])expect(replay(calls=>change(calls[1]!,from!,to!)),to).toBe(true); + expect(replay(calls=>calls[1]!.questions[0]!.options.reverse())).toBe(true); + expect(replay(calls=>{const c=calls[1]!;change(c,'Net: hand off now as you asked, or chain the eng review here.','Net: hand off now as you asked, or chain the eng review here.\n> Historical example: add a new task before leaving.');})).toBe(true); + expect(replay(calls=>{const o=calls[1]!.questions[0]!.options[0]!;o.description=o.description!.replace('Plan exits now with all DX decisions and tasks recorded; nothing else is started.','Exit the plan now with all DX tasks and decisions recorded. No further review is started.');})).toBe(true); + }); + test('source, conditional or withdrawn completion facts cannot make a stale report current',()=>{ + const edits:Array<[string,string]>=[ + ['D13 — DX review','Source: D13 — DX review'], + ['DX review complete','DX review is not complete'], + ['DX review complete','DX review complete only after another decision'], + ['ELI10: The DX review found','ELI10: Earlier review assessment: The DX review found'], + ['ELI10: The DX review found','ELI10: If approved, the DX review found'], + ['All are written into the plan as tasks T1 to T9.','Example: All are written into the plan as tasks T1 to T9.'], + ['All are written into the plan as tasks T1 to T9.','Previously, all are written into the plan as tasks T1 to T9.'], + ['All are written into the plan as tasks T1 to T9.','"All are written into the plan as tasks T1 to T9."'], + ['All are written into the plan as tasks T1 to T9.','All will be written into the plan as tasks T1 to T9.'], + ['Project/branch/task:','Source:\nProject/branch/task:'], + ['Project/branch/task: ','Project/branch/task: If approved, '], + ['Project/branch/task: ','Project/branch/task: Source excerpt, not a current assessment: '], + ]; + for(const [from,to] of edits)expect(replay(calls=>change(calls[1]!,from,to)),to).toBe(false); + for(const suffix of [' This review is not complete.',' These tasks are not recorded.',' This review is "withdrawn".',' One DX decision remains unresolved.',' We must fix another issue.',' Add another migration task.',' Should we approve another change?',' ']) { + expect(replay(calls=>{const c=calls[1]!;change(c,c.questions[0]!.question,c.questions[0]!.question+suffix);}),suffix).toBe(false); + } + }); + test('selected manual action must close with recorded decisions and no new work',()=>{ + for(const prefix of ['Source: ','Earlier review assessment: ','If approved, ','> '])expect(replay(calls=>{const o=calls[1]!.questions[0]!.options[0]!;o.description=prefix+o.description;}),prefix).toBe(false); + for(const suffix of [' Also update the plan before exit.',' Run /plan-eng-review now.',' This plan is not complete.',' The tasks are "withdrawn".',' This manual handoff is cancelled.'])expect(replay(calls=>{calls[1]!.questions[0]!.options[0]!.description+=suffix;}),suffix).toBe(false); + for(const from of ['all DX decisions and tasks recorded','nothing else is started'])expect(replay(calls=>{const o=calls[1]!.questions[0]!.options[0]!;o.description=o.description!.replace(from,'more work remains');}),from).toBe(false); + for(const index of [1,2])expect(replay(calls=>{const c=calls[1]!,q=c.questions[0]!;c.answers={[q.question]:q.options[index]!.label};})).toBe(false); + }); + test('completed owned native answer identity remains mandatory',()=>{ + const mutations:Array<(c:NativePlanQuestionCall)=>void>=[ + c=>{c.answered=false;},c=>{c.failed=true;},c=>{c.sessionId='foreign';},c=>{c.toolUseId='';}, + c=>{c.answeredAt='invalid';},c=>{c.answeredAt=new Date(Date.now()+60_000).toISOString();}, + c=>{c.unansweredQuestionIndices=[0];},c=>{delete c.unansweredQuestionIndices;}, + c=>{c.questions[0]!.multiSelect=true;},c=>{c.questions[0]!.header='Issue decision';}, + c=>{c.questions.push(structuredClone(c.questions[0]!));}, + c=>{c.answers={wrong:c.questions[0]!.options[0]!.label};}, + c=>{c.answers![c.questions[0]!.question]='Not an offered answer';}, + c=>{c.answers!.extra='foreign';}, + c=>{c.questions[0]!.options[1]!.label=c.questions[0]!.options[0]!.label;}, + ]; + for(const edit of mutations)expect(replay(calls=>edit(calls[1]!)),edit.toString()).toBe(false); + }); + test('other modifying answers and complete report/current Exit gates remain unchanged',()=>{ + expect(replay(calls=>{calls[0]!.answeredAt=new Date(captured.provenance.reportMtimeMs+1).toISOString();})).toBe(false); + expect(replay((_calls,_t,report)=>{const time=Date.parse(captured.calls[0]!.answeredAt!)/1000-1;fs.utimesSync(report,time,time);})).toBe(false); + expect(replay((_calls,_t,report)=>fs.writeFileSync(report,'# Completion summary\nDone.'))).toBe(false); + expect(replay((_calls,_t,report)=>fs.unlinkSync(report))).toBe(false); + for(const mutate of [ + (t:PlanCountTranscript)=>{t.planReadyRequests=[];}, + (t:PlanCountTranscript)=>{t.planReadyRequests![0]!.failed=true;}, + (t:PlanCountTranscript)=>{t.planReadyRequests![0]!.sessionId='foreign';}, + (t:PlanCountTranscript)=>{t.planReadyRequests![0]!.timestamp=captured.calls[1]!.answeredAt!;}, + (t:PlanCountTranscript)=>{t.status='missing';}, + ])expect(replay((_calls,t)=>mutate(t))).toBe(false); + }); +}); +}); + +describe('plan-count-dx-handoff-o', () => { +const fixture = fixture_plan_count_dx_handoff_o; +function replay(mutate?: (call: NativePlanQuestionCall, transcript: PlanCountTranscript) => void): boolean { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'dx-native-handoff-')); + const report = path.join(dir, 'review.md'); + const transcript = structuredClone(fixture.transcript) as PlanCountTranscript; + const call = transcript.calls.at(-1)!; + mutate?.(call, transcript); + fs.writeFileSync(report, fixture.report); + const writtenAt = Date.parse(fixture.reportWrite.resultAt); + fs.utimesSync(report, writtenAt / 1000, writtenAt / 1000); + try { + return hasNativePlanTerminal(transcript, report, + Date.parse(fixture.startedAt), 'plan_ready'); + } finally { + fs.rmSync(dir, { recursive: true, force: true }); + } +} + +test('actual completed DX navigation preserves the prior full report and all thirteen native calls', () => { + expect(fixture.transcript.calls).toHaveLength(13); + expect(fixture.transcript.calls.every(call => call.answered && !call.failed)).toBe(true); + expect(fixture.reportWrite.failed).toBe(false); + expect(replay()).toBe(true); +}); + +test('closed navigation cannot hide a real new issue, unfinished review, or conditional closure', () => { + const variants = [ + (q: any) => { q.header = 'New finding'; }, + (q: any) => { q.question = q.question.replace('D11 — What next?', 'D11 — Should we fix the missing authorization check?'); }, + (q: any) => { q.question = q.question.replace('The DX review is done.', 'The DX review is not done.'); }, + (q: any) => { q.question = q.question.replace('The DX review is done.', 'The DX review is done only after tests pass.'); }, + (q: any) => { q.question += '\nOne documentation gap remains unresolved.'; }, + (q: any) => { q.question += '\nPlease fix the missing authorization check first.'; }, + (q: any) => { q.question += '\nShould we add the missing test?'; }, + (q: any) => { q.question += '\nOnce the tests pass, the DX review will be complete.'; }, + (q: any) => { q.question += '\nAll decisions resolved after tests pass.'; }, + (q: any) => { q.question += '\nDX review complete after tests pass.'; }, + (q: any) => { q.question += '\nWe could fix the missing authorization check before Eng.'; }, + (q: any) => { q.question = q.question.replace('ELI10: The DX review is done.', '```text\nThe DX review is done.\n```'); }, + (q: any) => { q.question = q.question.replace('ELI10: The DX review is done.', '```text\nThe DX review is done.'); }, + (q: any) => { q.question = q.question.replace('ELI10: The DX review is done.', '> The DX review is done.'); }, + (q: any) => { q.options[2].description = 'Proceed to fix the missing contrast test before Eng.'; }, + (q: any) => { q.options[2].description = 'One issue remains unresolved; handle it manually.'; }, + (q: any) => { q.options[2].label = 'C) Add the missing test'; }, + (q: any) => { q.options.push({ label: 'D) Add a migration guide', description: 'A new required deliverable.' }); }, + (q: any) => { q.question = q.question.replace('plan-devex-review-next-steps', 'plan-devex-review-new-issue'); }, + (q: any) => { q.question += '\n { q.multiSelect = true; }, + ]; + for (const mutate of variants) { + expect(replay(call => { + const question = call.questions[0]!; + mutate(question); + // Keep a real offered answer after text mutations so the semantic guard, + // rather than a stale answer key, is what must reject the altered call. + call.answers = { [question.question]: question.options[0]!.label }; + }), String(mutate)).toBe(false); + } +}); + +test('native failure, offered answer, report freshness and real Exit remain required', () => { + expect(replay(call => { call.answered = false; })).toBe(false); + expect(replay(call => { call.failed = true; })).toBe(false); + expect(replay(call => { call.unansweredQuestionIndices = [0]; })).toBe(false); + expect(replay(call => { call.answers = { [call.questions[0]!.question]: 'Add a new test first' }; })).toBe(false); + expect(replay((_call, transcript) => { transcript.planReadyRequests = []; })).toBe(false); + expect(replay((_call, transcript) => { transcript.planReadyRequests![0]!.failed = true; })).toBe(false); + expect(replay((_call, transcript) => { transcript.planReadyRequests![0]!.sessionId = 'foreign'; })).toBe(false); + expect(replay((_call, transcript) => { + transcript.calls[11]!.answeredAt = new Date(Date.parse(fixture.reportWrite.resultAt) + 1000).toISOString(); + })).toBe(false); +}); + +test('pure navigation preserves actual option order and ordinary next-review sequencing', () => { + expect(replay(call => { call.questions[0]!.options.reverse(); })).toBe(true); + expect(replay(call => { + const q = call.questions[0]!; + q.options[0]!.description += ' After Eng review is complete, proceed to implementation.'; + })).toBe(true); +}); +}); + +describe('eng-next-handoff-ah', () => { +const fs = fs_ceo_handoff_y; +const os = os_ceo_handoff_y; +const path = path_ceo_handoff_y; +const actual = actual_eng_next_handoff_ah; +test('exact final exit/report replay retains all freshness, identity and answer gates', () => { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'gstack-eng-next-ah-')); + const file = path.join(dir, 'reviewed.md'); + const now = Date.now; + try { + fs.writeFileSync(file, actual.plan); + fs.utimesSync(file, actual.source.stat.mtimeMs / 1000, actual.source.stat.mtimeMs / 1000); + Date.now = () => Date.parse(actual.captureAt); + const t = structuredClone(actual.transcript) as PlanCountTranscript; + const id = actual.fingerprint.signature; + const admin = new Set([id]); + const check = (v = t, a = admin) => hasNativePlanTerminal(v, file, actual.startedAt, 'plan_ready', a); + expect(isCurrentPlanApprovalScreen(actual.screen)).toBe(true); + expect(check()).toBe(true); + expect(check(t, new Set())).toBe(false); + expect(check(t, new Set(['foreign:call']))).toBe(false); + for (const mutate of [ + (v: PlanCountTranscript) => { v.planReadyRequests = []; }, + (v: PlanCountTranscript) => { v.planReadyRequests!.at(-1)!.failed = true; }, + (v: PlanCountTranscript) => { v.planReadyRequests!.at(-1)!.sessionId = 'foreign'; }, + (v: PlanCountTranscript) => { v.planReadyRequests!.at(-1)!.timestamp = '2026-09-10T03:29:40.000Z'; }, + (v: PlanCountTranscript) => { v.planReadyRequests!.at(-1)!.timestamp = new Date(Date.now() + 1).toISOString(); }, + (v: PlanCountTranscript) => { v.calls.at(-1)!.answered = false; }, + (v: PlanCountTranscript) => { v.calls.at(-2)!.answeredAt = '2026-09-10T03:29:00.000Z'; }, + ]) { const v = structuredClone(t); mutate(v); expect(check(v)).toBe(false); } + fs.writeFileSync(file, actual.plan.replace('NO UNRESOLVED DECISIONS', 'Report still pending')); + fs.utimesSync(file, actual.source.stat.mtimeMs / 1000, actual.source.stat.mtimeMs / 1000); + expect(check()).toBe(false); + } finally { Date.now = now; fs.rmSync(dir, { recursive: true, force: true }); } +}); + +const b176 = actual.sourceBoundB176; +const recorded = () => structuredClone(b176.transcript) as PlanCountTranscript; +test('actual pending ExitPlanMode needs the classified recap plus the unchanged fresh report and native gates',()=>{ + const dir=fs.mkdtempSync(path.join(os.tmpdir(),'eng-b176-terminal-')),file=path.join(dir,'report.md'),now=Date.now; + try{ + Date.now=()=>Date.parse(b176.capturedAt);fs.writeFileSync(file,b176.plan);fs.utimesSync(file,b176.sourceReport.mtimeMs/1000,b176.sourceReport.mtimeMs/1000); + const t=recorded(),signature=b176.fingerprint.signature; + const admin=new Set([signature]); + const check=(transcript=t,administrative=admin)=>hasNativePlanTerminal(transcript,file,b176.startedAt,'plan_ready',administrative); + expect(isCurrentPlanApprovalScreen(b176.screen)).toBe(true); + expect(check()).toBe(true);expect(check(t,new Set())).toBe(false);expect(check(t,new Set(['foreign:call']))).toBe(false); + for(const mutate of [ + (v:PlanCountTranscript)=>{v.planReadyRequests=[];}, + (v:PlanCountTranscript)=>{v.planReadyRequests![0]!.failed=true;}, + (v:PlanCountTranscript)=>{v.planReadyRequests![0]!.sessionId='foreign';}, + (v:PlanCountTranscript)=>{v.planReadyRequests![0]!.timestamp=new Date(Date.now()+1).toISOString();}, + (v:PlanCountTranscript)=>{v.planReadyRequests![0]!.timestamp=v.calls.at(-1)!.answeredAt!;}, + (v:PlanCountTranscript)=>{v.calls.at(-1)!.answered=false;}, + (v:PlanCountTranscript)=>{v.calls.at(-2)!.answeredAt=new Date(b176.sourceReport.mtimeMs+1).toISOString();}, + ]){const v=recorded();mutate(v);expect(check(v)).toBe(false);} + fs.utimesSync(file,(b176.startedAt-1)/1000,(b176.startedAt-1)/1000);expect(check()).toBe(false); + fs.writeFileSync(file,b176.plan.replace('NO UNRESOLVED DECISIONS','PENDING'));fs.utimesSync(file,b176.sourceReport.mtimeMs/1000,b176.sourceReport.mtimeMs/1000);expect(check()).toBe(false); + }finally{Date.now=now;fs.rmSync(dir,{recursive:true,force:true});} +}); +}); + +describe('eng-task-pause-navigation-f359', () => { +const fs = fs_ceo_handoff_y; +const os = os_ceo_handoff_y; +const path = path_ceo_handoff_y; +const capture = capture_eng_task_pause_navigation_f359; +const actual=()=>({call:structuredClone(capture.transcript.calls.at(-1)!) as NativePlanQuestionCall,prior:structuredClone(capture.transcript.calls.slice(0,-1)) as NativePlanQuestionCall[],plan:capture.plan}); +test('handoff alone never supplies a native terminal or refreshes modifying answers',()=>{ + const x=actual(),fp=nativePlanCallFingerprint(x.call,0,false),admin=new Set([fp.signature]); + const dir=fs.mkdtempSync(path.join(os.tmpdir(),'gstack-eng-task-pause-')),file=path.join(dir,'reviewed.md'),now=Date.now; + try { + fs.writeFileSync(file,x.plan);fs.utimesSync(file,capture.reportSource.mtimeMs/1000,capture.reportSource.mtimeMs/1000); + Date.now=()=>Date.parse('2026-09-16T07:04:00.000Z'); + const t=structuredClone(capture.transcript) as PlanCountTranscript; + const check=(v=t,a=admin)=>hasNativePlanTerminal(v,file,Date.parse('2026-09-16T06:40:00.000Z'),'plan_ready',a); + expect(check()).toBe(false); // Actual capture precedes the native exit. + // The later retained native exit is real; this is a gate replay, not a + // replacement verdict for the original paid timeout/failure. + expect(createHash('sha256').update(JSON.stringify(t.calls)).digest('hex')).toBe(capture.terminalCapture.callsSha256); + t.planReadyRequests=structuredClone(capture.terminalCapture.planReadyRequests); + t.assistantMessages=structuredClone(capture.terminalCapture.assistantMessages); + expect(check()).toBe(true);expect(check(t,new Set())).toBe(false);expect(check(t,new Set(['foreign:call']))).toBe(false); + for(const mutate of [ + (v:PlanCountTranscript)=>{v.planReadyRequests![0]!.failed=true;}, + (v:PlanCountTranscript)=>{v.planReadyRequests![0]!.sessionId='foreign';}, + (v:PlanCountTranscript)=>{v.planReadyRequests![0]!.timestamp=x.call.answeredAt!;}, + (v:PlanCountTranscript)=>{v.planReadyRequests=[];}, + (v:PlanCountTranscript)=>{v.calls[5]!.answeredAt=x.call.answeredAt;}, + (v:PlanCountTranscript)=>{v.calls[5]!.answered=false;v.calls[5]!.unansweredQuestionIndices=[0];}, + ]){const v=structuredClone(t);mutate(v);expect(check(v)).toBe(false);} + }finally{Date.now=now;fs.rmSync(dir,{recursive:true,force:true});} +}); +}); + +describe('design-count-native-8525', () => { +const fixture = fixture_design_count_native_8525; +function completion() { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'design-8525-replay-')); + const file = path.join(dir, path.basename(fixture.provenance.planPath)); + const transcript = structuredClone(fixture.transcript) as PlanCountTranscript; + const edit = fixture.provenance.operations.filter(o => o.tool === 'Edit').at(-1)!; + const mtime = Date.parse(edit.acknowledgedAt) / 1000; + const write = (content = fixture.report) => { fs.writeFileSync(file, content); fs.utimesSync(file, mtime, mtime); }; + write(); + // The replay starts before the first retained native assistant message. + const startedAt = Math.min(...transcript.assistantMessages.map(m => Date.parse(m.timestamp))) - 1_000; + const final = transcript.assistantMessages.at(-1)!; + const check = () => hasNativePlanTerminal(transcript, file, startedAt, 'completion_summary'); + return { dir, file, transcript, final, write, check, cleanup: () => fs.rmSync(dir, {recursive:true, force:true}) }; +} + +test('exact native final text and reconstructed read-back-verified report supply completion', () => { + const f = completion(); try { expect(f.check()).toBe(true); } finally { f.cleanup(); } +}); +test('current typed status accepts presentation, field order and current report prose independently', () => { + const f=completion();try { + for (const heading of ['## Completion','### Completion summary','## Review complete','## Design review complete','**Review completion:**']) { + for (const status of ['STATUS: DONE','**STATUS:** DONE — review saved and verified.','**STATUS: DONE**']) { + for (const fields of [ + [status,`What changed: \`${path.basename(f.file)}\` now carries the current review report.`], + [`Report: ${f.file} contains the reviewed plan and verification.`,status], + [status,`- Plan saved to \`${f.file}\`.`], + ]) {f.final.text=heading+'\n\n'+fields.join('\n\n');expect(f.check(),f.final.text).toBe(true);} + } + } + } finally {f.cleanup();} +}); +for (const [name, change] of Object.entries({ + 'blocked':(s:string)=>s.replace('DONE —','BLOCKED —'), + 'concerns':(s:string)=>s.replace('DONE —','DONE_WITH_CONCERNS —'), + 'pending':(s:string)=>s.replace('DONE —','NEEDS_CONTEXT —'), + 'conditional status':(s:string)=>s.replace('DONE —','DONE if approved —'), + 'conditional reason':(s:string)=>s.replace('completed with evidence','will be completed with evidence'), + 'quoted status':(s:string)=>s.replace('**STATUS:**','> **STATUS:**'), + 'literal status':(s:string)=>s.replace(/\*\*STATUS:\*\* (.+)/,'`STATUS: $1`'), + 'fenced status':(s:string)=>s.replace(/\*\*STATUS:\*\* (.+)/,'```text\nSTATUS: $1\n```'), + 'duplicate status':(s:string)=>s+'\nSTATUS: DONE', + 'conflicting status':(s:string)=>s+'\nSTATUS: BLOCKED', + 'historical context':(s:string)=>'Previous result:\n\n'+s, + 'copied section':(s:string)=>'Source example:\n\n'+s, + 'quoted section':(s:string)=>'> '+s.replaceAll('\n','\n> '), + 'duplicate section':(s:string)=>s+'\n## Review complete\nSTATUS: DONE', + 'unavailable report':(s:string)=>s.replace('now carries','is unavailable; would contain'), + 'proposed write':(s:string)=>s.replace('now carries','will contain'), + 'historical report':(s:string)=>s.replace('now carries','previously contained'), + 'wrong path':(s:string)=>s.replaceAll('gstack-test-plan-design.md','wrong-plan.md'), + 'ambiguous path':(s:string)=>s.replace('now carries','and `another-plan.md` now carry'), + 'different absolute directory':(s:string)=>s.replaceAll('gstack-test-plan-design.md','/elsewhere/gstack-test-plan-design.md'), + 'relative traversal':(s:string)=>s.replaceAll('gstack-test-plan-design.md','../gstack-test-plan-design.md'), + 'quoted artifact line':(s:string)=>s.replace('**What changed:**','> **What changed:**'), + 'literal artifact prose':(s:string)=>s.replace(/\*\*What changed:\*\* (.+)/,'**What changed:** "$1"'), + 'missing artifact field':(s:string)=>s.replace(/^\*\*What changed:\*\*.+\n/m,''), + 'withdrawn report':(s:string)=>s+'\nThe report is withdrawn.', + 'remaining decision':(s:string)=>s+'\nOne design decision is unresolved.', +})) test(`typed delivery rejects ${name}`, () => {const f=completion();try {f.final.text=change(f.final.text);expect(f.check()).toBe(false);}finally{f.cleanup();}}); +test('typed delivery retains source session, answer chronology, fresh file and complete Design report checks', () => { + const f=completion();try { + const original=structuredClone(f.transcript); + for (const change of [ + (t:PlanCountTranscript)=>{t.calls[1]!.answered=false;}, + (t:PlanCountTranscript)=>{t.calls[1]!.failed=true;}, + (t:PlanCountTranscript)=>{t.calls[1]!.sessionId='foreign';}, + (t:PlanCountTranscript)=>{t.calls[1]!.answeredAt=t.assistantMessages.at(-1)!.timestamp;}, + (t:PlanCountTranscript)=>{t.assistantMessages.at(-1)!.timestamp='2999-01-01T00:00:00Z';}, + ]) {Object.assign(f.transcript,structuredClone(original));change(f.transcript);expect(f.check()).toBe(false);} + Object.assign(f.transcript,structuredClone(original)); + for (const body of ['# Draft',fixture.report+'\n## Implementation changes\n',fixture.report.replace('| 1 | clean |','| 1 | pending |'),fixture.report.replace('DESIGN CLEARED','NOT CLEARED'),fixture.report.replace('NO UNRESOLVED DECISIONS','**UNRESOLVED DECISIONS:**\n- Still open')]) {f.write(body);expect(f.check()).toBe(false);} + f.write();fs.utimesSync(f.file,1,1);expect(f.check()).toBe(false); + fs.rmSync(f.file);expect(f.check()).toBe(false); + const alternate=path.join(f.dir,'alternate.md');fs.writeFileSync(alternate,fixture.report);fs.symlinkSync(alternate,f.file);expect(f.check()).toBe(false); + }finally{f.cleanup();} +}); +const cf74 = fixture.cf74Retry; + +function cf74Completion() { + const dir=fs.mkdtempSync(path.join(os.tmpdir(),'design-cf74-completion-')); + const file=path.join(dir,path.basename(cf74.provenance.planPath)); + const transcript=structuredClone(cf74.transcript) as PlanCountTranscript; + const final=transcript.assistantMessages.at(-1)!; + final.text=final.text.replaceAll(cf74.provenance.planPath,file); + const startedAt=Math.min(...transcript.calls.map(c=>Date.parse(c.answeredAt!)))-1000; + const write=(body=cf74.report)=>{fs.writeFileSync(file,body);fs.utimesSync(file,cf74.provenance.reportMtimeMs/1000,cf74.provenance.reportMtimeMs/1000);}; + write(); + return {dir,file,transcript,final,startedAt,write,check:()=>hasNativePlanTerminal(transcript,file,startedAt,'completion_summary'),cleanup:()=>fs.rmSync(dir,{recursive:true,force:true})}; +} + +test('cf74 actual completed native report envelope binds the fresh owned Design report',()=>{ + const f=cf74Completion();try{expect(f.check()).toBe(true);}finally{f.cleanup();} +}); +for(const heading of ['## Completion report','### Completion summary','## Completion'])for(const field of ['Plan written:','Plan saved:','Plan written to']) + test(`cf74 complete typed delivery: ${heading}/${field}`,()=>{ + const f=cf74Completion();try{f.final.text=f.final.text.replace('## Completion report',heading).replace('Plan written:',field);expect(f.check()).toBe(true);}finally{f.cleanup();} + }); +for(const [name,change]of Object.entries({ + 'pending status':(s:string)=>s.replace('STATUS: DONE','STATUS: PENDING'), + 'conditional status':(s:string)=>s.replace('STATUS: DONE','STATUS: DONE if approved'), + 'quoted status':(s:string)=>s.replace('**STATUS: DONE**','`STATUS: DONE`'), + 'duplicate status':(s:string)=>s+'\nSTATUS: DONE', + 'quoted whole report':(s:string)=>'> '+s.replaceAll('\n','\n> '), + 'historical report':(s:string)=>s.replace('## Completion report','Historical source:\n\n## Completion report'), + 'duplicate report':(s:string)=>s+'\n## Completion report\nSTATUS: DONE', + 'future write':(s:string)=>s.replace('Plan written:','Plan will be written:'), + 'conditional write':(s:string)=>s.replace('Plan written:', 'Plan written if approved:'), + 'quoted written field':(s:string)=>s.replace('- **Plan written:**','> **Plan written:**'), + 'ambiguous path':(s:string)=>s.replace(' — accepted behavior',' and another-report.md — accepted behavior'), + 'foreign path':(s:string)=>s.replaceAll('gstack-test-plan-design.md','foreign-report.md'), + 'withdrawn report':(s:string)=>s+'\nThe review report is withdrawn.', + 'unresolved decision':(s:string)=>s+'\nOne design decision is unresolved.', + 'quoted current unresolved status':(s:string)=>s+'\nOne design decision is "unresolved".', +}))test(`cf74 typed completion rejects ${name}`,()=>{const f=cf74Completion();try{f.final.text=change(f.final.text);expect(f.check()).toBe(false);}finally{f.cleanup();}}); +test('cf74 typed envelope cannot bypass fresh own Design report and native chronology',()=>{ + const f=cf74Completion();try{ + const base=structuredClone(f.transcript); + for(const change of [ + (t:PlanCountTranscript)=>{t.calls[0]!.answered=false;},(t:PlanCountTranscript)=>{t.calls[0]!.failed=true;}, + (t:PlanCountTranscript)=>{t.calls[0]!.answers={};},(t:PlanCountTranscript)=>{t.calls[0]!.unansweredQuestionIndices=[0];}, + (t:PlanCountTranscript)=>{t.calls[0]!.sessionId='foreign';},(t:PlanCountTranscript)=>{t.calls[0]!.answeredAt=t.assistantMessages.at(-1)!.timestamp;}, + ]){Object.assign(f.transcript,structuredClone(base));change(f.transcript);expect(f.check()).toBe(false);} + Object.assign(f.transcript,structuredClone(base)); + for(const report of [cf74.report.replace('| 1 | clean |','| 1 | pending |'),cf74.report.replace('DESIGN CLEARED','DESIGN NOT CLEARED'),cf74.report.replace('NO UNRESOLVED DECISIONS','**UNRESOLVED DECISIONS:**\n- One pending'),cf74.report+'\n## Another section\n', '# Draft']){f.write(report);expect(f.check()).toBe(false);} + f.write();fs.utimesSync(f.file,1,1);expect(f.check()).toBe(false); + fs.rmSync(f.file);expect(f.check()).toBe(false); + const target=path.join(f.dir,'other.md');fs.writeFileSync(target,cf74.report);fs.symlinkSync(target,f.file);expect(f.check()).toBe(false); + }finally{f.cleanup();} +}); +}); diff --git a/test/plan-count-crop-ak.test.ts b/test/plan-count-crop-ak.test.ts deleted file mode 100644 index 190de38cd..000000000 --- a/test/plan-count-crop-ak.test.ts +++ /dev/null @@ -1,78 +0,0 @@ -import { expect, test } from 'bun:test'; -import fs from 'node:fs'; -import os from 'node:os'; -import path from 'node:path'; -import fixture from './fixtures/plan-count-crop-ak.json'; -import { currentFilePermissionEpoch } from './helpers/plan-count-file-permission'; -import { createPlanCountPermissionGuard } from './helpers/claude-pty-runner'; -import { E2E_TOUCHFILES, selectTests } from './helpers/touchfiles'; - -function replay(change: (f: any) => void = () => {}) { - const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'plan-crop-ak-')); - const expected = path.join(dir, 'report.md'), record = path.join(dir, 'record.json'); - const f: any = { expected, record, screen: fixture.screen.replaceAll(path.dirname(fixture.hook.expected), dir) - .replaceAll(path.basename(fixture.hook.expected), 'report.md'), state: { ...structuredClone(fixture.hook), expected }, - before: fixture.ownedBefore, cwd: fixture.cwd, config: fixture.config, startedAt: fixture.startedAt, - transcript: structuredClone(fixture.transcript), fileKind: 'file' }; - try { - change(f); fs.writeFileSync(record, JSON.stringify(f.state)); - if (f.fileKind === 'file') fs.writeFileSync(expected, f.before); - if (f.fileKind === 'directory') fs.mkdirSync(expected); - if (f.fileKind === 'symlink') { const other = path.join(dir, 'other.md'); fs.writeFileSync(other, f.before); fs.symlinkSync(other, expected); } - const epoch = currentFilePermissionEpoch(record, f.expected, f.cwd, f.config, f.startedAt, f.transcript, f.screen); - return { epoch, screen: f.screen, guard: createPlanCountPermissionGuard(), run: () => createPlanCountPermissionGuard()(f.screen, '', epoch) }; - } finally { fs.rmSync(dir, { recursive: true, force: true }); } -} - -test('actual owned bare continuation uses the preceding original line and grants once', () => { - const r = replay(); - expect(r.epoch?.pendingId).toBe(fixture.hook.pendingId); - expect(r.guard(r.screen, '', r.epoch)).toBe('grant'); - expect(r.guard(r.screen, '', r.epoch)).toBe('handled'); -}); - -test('harmless preceding line number relocation keeps the exact content binding', () => { - const r = replay(f => { - f.before = 'Added unchanged line\n' + f.before; - f.screen = f.screen.replace(/^ {0,3}82 +$/m, ' 83 '); - }); - expect(r.epoch?.pendingId).toBe(fixture.hook.pendingId); -}); - -const negatives: Array<[string, (f: any) => void]> = [ - ['wrong preceding line', f => { f.screen = f.screen.replace(/^ {0,3}82 +$/m, ' 83 '); }], - ['unrelated prose continuation', f => { f.screen = f.screen.replace(/^.*\n/, ' Apply this edit now\n'); }], - ['source quotation continuation', f => { f.screen = f.screen.replace(/^.*\n/, ' > Example: apply this edit now\n'); }], - ['foreign continuation suffix', f => { f.screen = f.screen.replace('f2 (~7.6:1)', 'foreign (~7.6:1)'); }], - ['five-space unnumbered gutter', f => { f.screen = f.screen.slice(1); }], - ['seven-space unnumbered gutter', f => { f.screen = ' ' + f.screen; }], - ['no following unchanged numbered row', f => { f.screen = f.screen.replace(/^ {0,3}82 +$/m, ' 82 +'); }], - ['two arbitrary continuation rows', f => { f.screen = f.screen.split('\n')[0] + '\n' + f.screen; }], - ['missing original file', f => { f.fileKind = 'missing'; }], - ['directory in place of original file', f => { f.fileKind = 'directory'; }], - ['required original line beyond the bounded prefix', f => { f.before = 'x'.repeat(65537) + f.before; }], - ['foreign displayed directory', f => { f.screen = f.screen.replace(path.dirname(f.expected) + ' for this session', path.join(path.dirname(f.expected), 'foreign') + ' for this session'); }], - ['foreign hook expected path', f => { f.state.expected += '.foreign'; }], - ['no current native request', f => { f.state.pendingId = null; }], - ['completed native request', f => { f.state.completedId = f.state.pendingId; }], - ['stale native timestamp', f => { f.state.timestamp = new Date(f.startedAt - 1).toISOString(); }], - ['future native timestamp', f => { f.state.timestamp = new Date(Date.now() + 60000).toISOString(); }], - ['foreign session transcript', f => { f.transcript.assistantMessages[0].sessionId = 'foreign'; }], - ['unavailable native transcript', f => { f.transcript.status = 'error'; }], - ['missing complete menu footer', f => { f.screen = f.screen.replace('Esc to cancel · Tab to amend', ''); }], - ['selected policy change', f => { f.screen = f.screen.replace('❯ 1. Yes', '❯ 1. Yes, always allow'); }], - ['unrelated prompt prepended', f => { f.screen = '☐ Review this example\n' + f.screen; }], - ['historical pane is not current viewport', f => { f.screen = 'Waiting for current tool'; }], -]; -test.each(negatives)('%s cannot supply an owned epoch', (_, change) => { - const r = replay(change); expect(r.epoch).not.toBeTruthy(); expect(r.run()).not.toBe('grant'); -}); -test.skipIf(process.platform === 'win32')('a symlink cannot supply current original content', () => { - expect(replay(f => { f.fileKind = 'symlink'; }).epoch).toBeNull(); -}); -test('new dependencies have the exact existing file-permission consumers', () => { - for (const dependency of ['test/plan-count-crop-ak.test.ts', 'test/fixtures/plan-count-crop-ak.json']) { - expect(selectTests([dependency], E2E_TOUCHFILES).selected.sort()).toEqual( - selectTests(['test/helpers/plan-count-file-permission.ts'], E2E_TOUCHFILES).selected.sort()); - } -}); diff --git a/test/plan-count-dx-handoff-o.test.ts b/test/plan-count-dx-handoff-o.test.ts deleted file mode 100644 index a3c173a73..000000000 --- a/test/plan-count-dx-handoff-o.test.ts +++ /dev/null @@ -1,87 +0,0 @@ -import { expect, test } from 'bun:test'; -import * as fs from 'node:fs'; -import * as os from 'node:os'; -import * as path from 'node:path'; -import { hasNativePlanTerminal } from './helpers/claude-pty-runner'; -import type { NativePlanQuestionCall, PlanCountTranscript } from './helpers/plan-count-transcript'; -import fixture from './fixtures/devex-handoff-o-call.json'; - -function replay(mutate?: (call: NativePlanQuestionCall, transcript: PlanCountTranscript) => void): boolean { - const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'dx-native-handoff-')); - const report = path.join(dir, 'review.md'); - const transcript = structuredClone(fixture.transcript) as PlanCountTranscript; - const call = transcript.calls.at(-1)!; - mutate?.(call, transcript); - fs.writeFileSync(report, fixture.report); - const writtenAt = Date.parse(fixture.reportWrite.resultAt); - fs.utimesSync(report, writtenAt / 1000, writtenAt / 1000); - try { - return hasNativePlanTerminal(transcript, report, - Date.parse(fixture.startedAt), 'plan_ready'); - } finally { - fs.rmSync(dir, { recursive: true, force: true }); - } -} - -test('actual completed DX navigation preserves the prior full report and all thirteen native calls', () => { - expect(fixture.transcript.calls).toHaveLength(13); - expect(fixture.transcript.calls.every(call => call.answered && !call.failed)).toBe(true); - expect(fixture.reportWrite.failed).toBe(false); - expect(replay()).toBe(true); -}); - -test('closed navigation cannot hide a real new issue, unfinished review, or conditional closure', () => { - const variants = [ - (q: any) => { q.header = 'New finding'; }, - (q: any) => { q.question = q.question.replace('D11 — What next?', 'D11 — Should we fix the missing authorization check?'); }, - (q: any) => { q.question = q.question.replace('The DX review is done.', 'The DX review is not done.'); }, - (q: any) => { q.question = q.question.replace('The DX review is done.', 'The DX review is done only after tests pass.'); }, - (q: any) => { q.question += '\nOne documentation gap remains unresolved.'; }, - (q: any) => { q.question += '\nPlease fix the missing authorization check first.'; }, - (q: any) => { q.question += '\nShould we add the missing test?'; }, - (q: any) => { q.question += '\nOnce the tests pass, the DX review will be complete.'; }, - (q: any) => { q.question += '\nAll decisions resolved after tests pass.'; }, - (q: any) => { q.question += '\nDX review complete after tests pass.'; }, - (q: any) => { q.question += '\nWe could fix the missing authorization check before Eng.'; }, - (q: any) => { q.question = q.question.replace('ELI10: The DX review is done.', '```text\nThe DX review is done.\n```'); }, - (q: any) => { q.question = q.question.replace('ELI10: The DX review is done.', '```text\nThe DX review is done.'); }, - (q: any) => { q.question = q.question.replace('ELI10: The DX review is done.', '> The DX review is done.'); }, - (q: any) => { q.options[2].description = 'Proceed to fix the missing contrast test before Eng.'; }, - (q: any) => { q.options[2].description = 'One issue remains unresolved; handle it manually.'; }, - (q: any) => { q.options[2].label = 'C) Add the missing test'; }, - (q: any) => { q.options.push({ label: 'D) Add a migration guide', description: 'A new required deliverable.' }); }, - (q: any) => { q.question = q.question.replace('plan-devex-review-next-steps', 'plan-devex-review-new-issue'); }, - (q: any) => { q.question += '\n { q.multiSelect = true; }, - ]; - for (const mutate of variants) { - expect(replay(call => { - const question = call.questions[0]!; - mutate(question); - // Keep a real offered answer after text mutations so the semantic guard, - // rather than a stale answer key, is what must reject the altered call. - call.answers = { [question.question]: question.options[0]!.label }; - }), String(mutate)).toBe(false); - } -}); - -test('native failure, offered answer, report freshness and real Exit remain required', () => { - expect(replay(call => { call.answered = false; })).toBe(false); - expect(replay(call => { call.failed = true; })).toBe(false); - expect(replay(call => { call.unansweredQuestionIndices = [0]; })).toBe(false); - expect(replay(call => { call.answers = { [call.questions[0]!.question]: 'Add a new test first' }; })).toBe(false); - expect(replay((_call, transcript) => { transcript.planReadyRequests = []; })).toBe(false); - expect(replay((_call, transcript) => { transcript.planReadyRequests![0]!.failed = true; })).toBe(false); - expect(replay((_call, transcript) => { transcript.planReadyRequests![0]!.sessionId = 'foreign'; })).toBe(false); - expect(replay((_call, transcript) => { - transcript.calls[11]!.answeredAt = new Date(Date.parse(fixture.reportWrite.resultAt) + 1000).toISOString(); - })).toBe(false); -}); - -test('pure navigation preserves actual option order and ordinary next-review sequencing', () => { - expect(replay(call => { call.questions[0]!.options.reverse(); })).toBe(true); - expect(replay(call => { - const q = call.questions[0]!; - q.options[0]!.description += ' After Eng review is complete, proceed to implementation.'; - })).toBe(true); -}); diff --git a/test/plan-count-file-permission.test.ts b/test/plan-count-file-permission.test.ts index 830de2aeb..e7047338d 100644 --- a/test/plan-count-file-permission.test.ts +++ b/test/plan-count-file-permission.test.ts @@ -207,7 +207,672 @@ test('AD v2 path-only heading must agree with full menu target and current metad }); import {E2E_TOUCHFILES,selectTests} from './helpers/touchfiles'; +import fs_batching_permission_at from 'node:fs'; +import os_batching_permission_at from 'node:os'; +import path_batching_permission_at from 'node:path'; +import captured_batching_permission_at from './fixtures/batching-permission-at.json'; +import fixture_design_crop_gutter_ap from './fixtures/design-crop-gutter-ap.json'; +import previous_design_crop_gutter_ap from './fixtures/plan-count-crop-ak.json'; +import capturedAd_plan_count_permission_ac from './fixtures/plan-count-permission-ad.json'; +import capturedAe_plan_count_permission_ac from './fixtures/plan-count-permission-ae.json'; +import capturedAh_plan_count_permission_ac from './fixtures/plan-count-permission-ah.json'; +import { currentFilePermissionBinding } from './helpers/plan-count-file-permission'; +import exact_plan_count_quoted_frame_ak from './fixtures/plan-count-quoted-frame-ak.json'; +import owned_plan_count_quoted_frame_ak from './fixtures/plan-count-owned-permission-v.json'; test('AD v2 cropped target fixture selects all existing file-permission consumers',()=>{ expect(selectTests(['test/fixtures/plan-count-permission-target-ad-v2.json'],E2E_TOUCHFILES,[]).selected) .toEqual(selectTests(['test/plan-count-file-permission.test.ts'],E2E_TOUCHFILES,[]).selected); }); + +describe('batching-permission-at', () => { +const fs = fs_batching_permission_at; +const os = os_batching_permission_at; +const path = path_batching_permission_at; +const captured = captured_batching_permission_at; +function renderPermissionScreen(expected: string, paths: Pick = path): string { + // The capture is already laid out at the runner's 120 columns. Replacing its + // path must reflow that menu line, otherwise the PTY hard-wraps words in half. + return captured.screen.split('\n').map(original => { + const line = original.replaceAll(path.posix.dirname(captured.expectedPath), paths.dirname(expected)) + .replaceAll(path.posix.basename(captured.expectedPath), paths.basename(expected)); + if (line === original || line.length <= 120) return line; + const indent = /^ */.exec(line)![0], lines: string[] = []; let current = indent; + for (const word of line.trim().split(/\s+/)) { + if (indent.length + word.length > 120) throw Error('Fixture path exceeds the permission panel width'); + if (current.length > indent.length && current.length + 1 + word.length > 120) { lines.push(current); current = indent; } + current += (current.length > indent.length ? ' ' : '') + word; + } + return [...lines, current].join('\n'); + }).join('\n'); +} + +function fixture(){ + const dir=fs.mkdtempSync(path.join(os.tmpdir(),'batch-permission-')),cwd=path.join(dir,'cwd'),config=path.join(dir,'.claude'),expected=path.join(dir,'report.md');fs.mkdirSync(cwd);fs.writeFileSync(expected,'original'); + const recorder=createFilePermissionRecorder(cwd,config,expected)!;const startedAt=Date.now()-1000; + const screen=renderPermissionScreen(expected); + const transcript:any={status:'ready',calls:[],assistantMessages:[{sessionId:'synthetic-epoch',text:'Reviewing',timestamp:new Date().toISOString()}]}; + const record=(name:string,id:string,extra={})=>recordFilePermission(JSON.stringify({hook_event_name:name,tool_name:'Edit',session_id:'synthetic-epoch',tool_use_id:id,cwd,transcript_path:path.join(config,'projects','owned','synthetic-epoch.jsonl'),tool_input:{file_path:expected},...extra}),recorder.file,cwd,config,expected); + const read=()=>currentFilePermissionEpoch(recorder.file,expected,cwd,config,startedAt,transcript,screen); + return{dir,cwd,config,expected,recorder,screen,transcript,record,read,close(){recorder.dispose();fs.rmSync(dir,{recursive:true,force:true})}}; +} + +test('retained retry has a valid permission panel and real previous completion without a pending ID',()=>{ + expect(classifyPlanCountFrame(captured.screen)).toBe('permission');expect(captured.priorCompletedEdit[0]!.name).toBe('Edit');expect(captured.priorCompletedEdit[1]!.isError).toBe(false); + expect(captured.pendingEditId).toBeNull();expect(captured.provenance.originalOutcome).toBe('timeout');expect(captured.provenance.paidOutcomeReclassified).toBe(false); + const guard=createPlanCountPermissionGuard();expect(guard(captured.screen,captured.lastMatchedDisplayCompletion)).toBe('grant');expect(guard(captured.screen,captured.lastMatchedDisplayCompletion)).toBe('handled'); +}); + +test('a substituted long fixture path reflows the menu without splitting permission words', () => { + const prefix = ' always allow access to ', suffix = ' for this '; + const directory = '/' + 'x'.repeat(120 - prefix.length - suffix.length - 3 - 1); + const rawLine = `${prefix}${directory}${suffix}session`; + expect(`${rawLine.slice(0, 120)}\n${rawLine.slice(120)}`).toContain('ses\nsion'); + for (const paths of [path.posix, path.win32]) { + const expected = paths.join(directory, 'report.md'); + const screen = renderPermissionScreen(expected, paths); + const menu = screen.slice(screen.indexOf(' Do you want to make this edit')); + expect(menu.split('\n').every(line => line.length <= 120)).toBe(true); + expect(menu).toContain(paths.dirname(expected)); + expect(menu).toContain('edit to report.md?'); + expect(menu).toMatch(/1\. Yes[\s\S]+2\. Yes,[\s\S]+3\. No/); + expect(createPlanCountPermissionGuard()(screen, captured.lastMatchedDisplayCompletion)).toBe('grant'); + } +}); + +test('synthetic hook epochs release only the later exact request after its predecessor succeeds',()=>{ + const f=fixture();try{const guard=createPlanCountPermissionGuard(),input=()=>guard(f.screen,captured.lastMatchedDisplayCompletion,f.read()); + expect(input()).toBe('handled');f.record('PreToolUse','first');expect(input()).toBe('grant');expect(input()).toBe('handled'); + f.record('PostToolUse','first');expect(input()).toBe('handled');f.record('PreToolUse','first');expect(input()).toBe('handled'); + f.record('PreToolUse','second');expect(input()).toBe('grant');expect(input()).toBe('handled');f.record('PostToolUse','first');expect(input()).toBe('handled'); + }finally{f.close()} +}); +for(const reason of ['failed','no-result','foreign-session','foreign-path','other-tool','sidechain'])test(`a later matching menu cannot replace ${reason} predecessor evidence`,()=>{ + const f=fixture();try{const guard=createPlanCountPermissionGuard(),input=()=>guard(f.screen,'',f.read());f.record('PreToolUse','first');expect(input()).toBe('grant'); + if(reason==='failed')f.record('PostToolUseFailure','first');else if(reason!=='no-result')f.record('PostToolUse','first',reason==='foreign-session'?{session_id:'foreign'}:reason==='foreign-path'?{tool_input:{file_path:path.join(f.dir,'foreign','report.md')}}:reason==='other-tool'?{tool_name:'Read'}:{agent_id:'child'}); + f.record('PreToolUse','second');expect(input()).toBe('handled'); + }finally{f.close()} +}); +test.skipIf(process.platform==='win32')('real fake CLI observes two file epochs without imposing terminal report validation',async()=>{ + const dir=fs.mkdtempSync(path.join(os.tmpdir(),'batch-permission-pty-')),fake=path.join(dir,'fake-claude'),worker=path.join(dir,'worker.ts'),events=path.join(dir,'events.jsonl'),output=path.join(dir,'output.json'),expected=path.join(dir,'report.md');fs.writeFileSync(expected,'original'); + const screen=renderPermissionScreen(expected); + fs.writeFileSync(fake,`#!${process.execPath}\n`+String.raw` +import * as fs from 'node:fs';import * as path from 'node:path'; +const item=JSON.parse(process.env.FILE_EPOCH_CASE);const log=e=>fs.appendFileSync(item.events,JSON.stringify(e)+'\n'); +const sid='epoch-main';const nativePath=path.join(process.env.CLAUDE_CONFIG_DIR,'projects','epoch',sid+'.jsonl');fs.mkdirSync(path.dirname(nativePath),{recursive:true}); +const native=(role,content,extra={})=>fs.appendFileSync(nativePath,JSON.stringify({cwd:process.cwd(),sessionId:sid,isSidechain:false,timestamp:new Date().toISOString(),message:{role,content},...extra})+'\n'); +native('assistant',[{type:'text',text:'Reviewing fixture.'}]);log({type:'start',pid:process.pid,cwd:process.cwd()}); +const settings=JSON.parse(process.argv[process.argv.indexOf('--settings')+1]); +if(settings.hooks.PreToolUse[0].matcher!=='^ExitPlanMode$')throw Error('Exit recorder changed'); +const hook=async(name,id)=>{ + const entries=(settings.hooks[name]??[]).filter(h=>h.matcher==='^(Write|Edit)$'); + if(entries.length!==1)throw Error('Expected exactly one caller-owned file recorder'); + for(const entry of entries){ + const event={hook_event_name:name,tool_name:'Edit',session_id:sid,tool_use_id:id,cwd:process.cwd(),transcript_path:nativePath,tool_input:{file_path:item.activePlan?path.join(process.cwd(),'PLAN.md'):item.expected,old_string:'old',new_string:'new'}}; + const p=Bun.spawn(['bash','-c',entry.hooks[0].command],{stdin:new Blob([JSON.stringify(event)]),stdout:'pipe',stderr:'pipe'}); + const [code,out,err]=await Promise.all([p.exited,new Response(p.stdout).text(),new Response(p.stderr).text()]);if(code||out||err)throw Error('hook was not silent');log({type:'hook',name,id}); + } +}; +let stage='startup';const paint=()=>process.stdout.write('\x1b[2J\x1b[H'+item.screen.replaceAll('__ACTIVE_PLAN_PATH__',path.join(process.cwd(),'PLAN.md')).replaceAll('\n','\r\n')); +process.stdin.setRawMode?.(true);process.stdin.on('data',async data=>{ + const input=data.toString();log({type:'input',stage,input}); + if(stage==='startup'){stage='first';await hook('PreToolUse','first');paint();return;} + if(stage==='old-pane'||stage==='done'){log({type:'unexpected'});return;} + if(input!=='1\r')throw Error('default permission input changed'); + if(stage==='first'){stage='old-pane';await hook('PostToolUse','first');if(item.intervening){await hook('PreToolUse','automatic');await hook('PostToolUse','automatic');}paint();setTimeout(async()=>{await hook('PreToolUse','second');stage='second';paint();},3200);return;} + stage='done';await hook('PostToolUse','second'); + const q={header:'Finding',question:'Apply this repair?',options:[{label:'Fix'},{label:'Keep'}]}; + native('assistant',[{type:'tool_use',name:'AskUserQuestion',id:'finding',input:{questions:[q]}}]);native('user',[{type:'tool_result',tool_use_id:'finding',content:'Answered'}],{toolUseResult:{answers:{[q.question]:'Fix'}}}); + process.stdout.write('\x1b[2J\x1b[HCompletion summary\r\n'); +});process.on('SIGINT',()=>process.exit(0));process.stdin.resume();process.stdout.write('FILE_EPOCH_READY\r\n'); +`);fs.chmodSync(fake,0o755); + fs.writeFileSync(worker,`import {runPlanSkillCounting} from ${JSON.stringify(pathToFileURL(path.join(import.meta.dir,'helpers/claude-pty-runner.ts')).href)};const o=await runPlanSkillCounting({skillName:'plan-eng-review',slashCommand:'/plan-eng-review',followUpPrompt:'Review this disposable batching fixture.',permissionPlanPath:${JSON.stringify(expected)},startupReadyMarker:'FILE_EPOCH_READY',isLastStep0AUQ:()=>false,isReviewAUQ:()=>true,reviewCountCeiling:2,timeoutMs:28000,env:{FILE_EPOCH_CASE:${JSON.stringify(JSON.stringify({events,expected,screen}))}}});await Bun.write(${JSON.stringify(output)},JSON.stringify(o));`); + const child=Bun.spawn([process.execPath,worker],{env:{...process.env,BROWSE_TERMINAL_BINARY:fake,EVALS_HERMETIC:'1'},stdout:'pipe',stderr:'pipe'});const killer=setTimeout(()=>child.kill('SIGKILL'),33000); + try{const[code,out,err]=await Promise.all([child.exited,new Response(child.stdout).text(),new Response(child.stderr).text()]);expect(code,out+err).toBe(0); + const o=JSON.parse(fs.readFileSync(output,'utf8'));expect(o.outcome,JSON.stringify(o)).toBe('completion_summary');expect(o.reviewCount).toBe(1);expect(fs.readFileSync(expected,'utf8')).toBe('original'); + const rows=fs.readFileSync(events,'utf8').trim().split('\n').map(l=>JSON.parse(l));expect(rows.filter(e=>e.type==='input').map(e=>e.input)).toEqual(['/plan-eng-review\r','1\r','1\r']);expect(rows.some(e=>e.type==='unexpected')).toBe(false); + expect(()=>process.kill(rows[0].pid,0)).toThrow();expect(fs.existsSync(rows[0].cwd)).toBe(false); + }finally{clearTimeout(killer);child.kill('SIGKILL');if(fs.existsSync(events)){const first=JSON.parse(fs.readFileSync(events,'utf8').split('\n')[0]!);try{process.kill(first.pid,'SIGKILL');}catch{}}fs.rmSync(dir,{recursive:true,force:true});} +},35000); +}); + +describe('design-crop-gutter-ap', () => { +const fs = fs_batching_permission_at; +const os = os_batching_permission_at; +const path = path_batching_permission_at; +const fixture = fixture_design_crop_gutter_ap; +const previous = previous_design_crop_gutter_ap; +function replay(change:(f:any)=>void=()=>{},input:any=fixture){ + const dir=fs.mkdtempSync(path.join(os.tmpdir(),'design-gutter-ap-')); + const expected=path.join(dir,'report.md'),record=path.join(dir,'record.json'); + const f:any={expected,record,cwd:input.cwd,config:input.config,startedAt:input.startedAt, + screen:input.screen.replaceAll(path.dirname(input.hook.expected),dir).replaceAll(path.basename(input.hook.expected),'report.md'), + state:{...structuredClone(input.hook),expected},before:input.ownedBefore,transcript:structuredClone(input.transcript),fileKind:'file'}; + try{ + change(f);fs.writeFileSync(record,JSON.stringify(f.state)); + if(f.fileKind==='file')fs.writeFileSync(expected,f.before); + if(f.fileKind==='directory')fs.mkdirSync(expected); + if(f.fileKind==='symlink'){const target=path.join(dir,'other.md');fs.writeFileSync(target,f.before);fs.symlinkSync(target,expected);} + const epoch=currentFilePermissionEpoch(record,f.expected,f.cwd,f.config,f.startedAt,f.transcript,f.screen); + const guard=createPlanCountPermissionGuard(); + return {epoch,first:guard(f.screen,'',epoch),second:guard(f.screen,'',epoch)}; + }finally{fs.rmSync(dir,{recursive:true,force:true});} +} + +test('exact five-column current gutter supplies one owned Edit epoch and one grant',()=>{ + const lines=fixture.screen.split('\n');expect(lines[0]).toMatch(/^ {5}\S/);expect(lines[1]).toBe(' 62 '); + expect(fixture.ownedBefore.split('\n')[60]!.endsWith(lines[0]!.slice(5))).toBe(true); + const r=replay();expect(r.epoch).toEqual({pendingId:fixture.hook.pendingId,completedId:fixture.hook.completedId,completedIds:fixture.hook.completedIds}); + expect(r.first).toBe('grant');expect(r.second).toBe('handled'); +}); + +test('prior six-column public crop remains exact and one-time',()=>{ + expect(previous.screen.split('\n')[0]).toMatch(/^ {6}\S/);expect(previous.screen.split('\n')[1]).toBe(' 82 '); + const r=replay(()=>{},previous);expect(r.epoch?.pendingId).toBe(previous.hook.pendingId);expect(r.first).toBe('grant');expect(r.second).toBe('handled'); + // Keep the existing six-space acceptance even when the adjacent numeric + // row has a different padding; the unchanged original-line guard remains. + expect(replay(f=>{f.screen=' '+f.screen;}).epoch?.pendingId).toBe(fixture.hook.pendingId); +}); + +test('padding and line-number width derive the continuation column together',()=>{ + const padded=replay(f=>{f.screen=' '+f.screen;f.screen=f.screen.replace(/^ 62 $/m,' 62 ');}); + expect(padded.epoch?.pendingId).toBe(fixture.hook.pendingId); + const relocated=replay(f=>{ + f.before='Earlier unchanged line\n'.repeat(38)+f.before; + f.screen=' '+f.screen; + f.screen=f.screen.replace(/^ ([1-9]\d*)( | [+-])/gm,(_:string,n:string,g:string)=>' '+(Number(n)+38)+g); + }); + expect(relocated.epoch?.pendingId).toBe(fixture.hook.pendingId); +}); + +test.each([0,3])('a native numbered-row padding of %d derives a matching non-six gutter',padding=>{ + const r=replay(f=>{ + f.screen=' '.repeat(padding+4)+f.screen.slice(5); + f.screen=f.screen.replace(/^ 62 $/m,' '.repeat(padding)+'62 '); + }); + expect(r.epoch?.pendingId).toBe(fixture.hook.pendingId);expect(r.first).toBe('grant'); +}); + +test('an ordinary numbered unchanged row remains a numbered row, not a wrapped continuation',()=>{ + const r=replay(f=>{ + f.screen=' 61 '+f.before.split('\n')[60]+'\n'+f.screen.slice(f.screen.indexOf('\n')+1); + }); + expect(r.epoch?.pendingId).toBe(fixture.hook.pendingId);expect(r.first).toBe('grant'); +}); + +const negatives:Array<[string,(f:any)=>void]>=[ + ['four-space gutter with five-column numbered row',f=>{f.screen=f.screen.slice(1);}], + ['seven-space gutter with five-column numbered row',f=>{f.screen=' '+f.screen;}], + ['tab cannot substitute for a native space gutter',f=>{f.screen='\t'+f.screen.slice(1);}], + ['wrong preceding file line',f=>{f.screen=f.screen.replace(/^ 62 $/m,' 63 ');}], + ['changed continuation content',f=>{f.screen=f.screen.replace('f2 with icon','foreign with icon');}], + ['stale before-file bytes',f=>{f.before=f.before.replace('f2 with icon','changed with icon');}], + ['quoted continuation',f=>{f.screen=f.screen.replace(/^ {5}/,' > ');}], + ['two unnumbered continuation rows',f=>{f.screen=f.screen.split('\n')[0]+'\n'+f.screen;}], + ['next row is an addition, not unchanged context',f=>{f.screen=f.screen.replace(/^ 62 $/m,' 62 +');}], + ['zero next line',f=>{f.screen=f.screen.replace(/^ 62 $/m,' 00 ');}], + ['missing current file',f=>{f.fileKind='missing';}], + ['directory instead of current file',f=>{f.fileKind='directory';}], + ['required source line beyond the bounded prefix',f=>{f.before='x'.repeat(65537)+f.before;}], + ['foreign displayed directory',f=>{f.screen=f.screen.replace(path.dirname(f.expected)+' for this session',path.join(path.dirname(f.expected),'foreign')+' for this session');}], + ['foreign hook target',f=>{f.state.expected+='.foreign';}], + ['foreign hook cwd',f=>{f.state.cwd+='.foreign';}], + ['foreign native session',f=>{f.transcript={status:'ready',calls:[],assistantMessages:[{sessionId:'foreign',text:'Current review',timestamp:new Date(f.startedAt).toISOString()}]};}], + ['missing pending request',f=>{f.state.pendingId=null;}], + ['completed request cannot reopen',f=>{f.state.completedId=f.state.pendingId;}], + ['stale request timestamp',f=>{f.state.timestamp=new Date(f.startedAt-1).toISOString();}], + ['missing menu footer',f=>{f.screen=f.screen.replace('Esc to cancel · Tab to amend','');}], + ['one-time action changed',f=>{f.screen=f.screen.replace('❯ 1. Yes','❯ 1. Yes, always allow');}], +]; +test.each(negatives)('%s cannot obtain a grant',(_,change)=>{ + const r=replay(change);expect(r.epoch).not.toBeTruthy();expect(r.first).not.toBe('grant'); +}); +test.skipIf(process.platform==='win32')('symlink cannot provide the original line',()=>{expect(replay(f=>{f.fileKind='symlink';}).epoch).toBeNull();}); +}); + +describe('plan-count-crop-ak', () => { +const fs = fs_batching_permission_at; +const os = os_batching_permission_at; +const path = path_batching_permission_at; +const fixture = previous_design_crop_gutter_ap; +function replay(change: (f: any) => void = () => {}) { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'plan-crop-ak-')); + const expected = path.join(dir, 'report.md'), record = path.join(dir, 'record.json'); + const f: any = { expected, record, screen: fixture.screen.replaceAll(path.dirname(fixture.hook.expected), dir) + .replaceAll(path.basename(fixture.hook.expected), 'report.md'), state: { ...structuredClone(fixture.hook), expected }, + before: fixture.ownedBefore, cwd: fixture.cwd, config: fixture.config, startedAt: fixture.startedAt, + transcript: structuredClone(fixture.transcript), fileKind: 'file' }; + try { + change(f); fs.writeFileSync(record, JSON.stringify(f.state)); + if (f.fileKind === 'file') fs.writeFileSync(expected, f.before); + if (f.fileKind === 'directory') fs.mkdirSync(expected); + if (f.fileKind === 'symlink') { const other = path.join(dir, 'other.md'); fs.writeFileSync(other, f.before); fs.symlinkSync(other, expected); } + const epoch = currentFilePermissionEpoch(record, f.expected, f.cwd, f.config, f.startedAt, f.transcript, f.screen); + return { epoch, screen: f.screen, guard: createPlanCountPermissionGuard(), run: () => createPlanCountPermissionGuard()(f.screen, '', epoch) }; + } finally { fs.rmSync(dir, { recursive: true, force: true }); } +} + +test('actual owned bare continuation uses the preceding original line and grants once', () => { + const r = replay(); + expect(r.epoch?.pendingId).toBe(fixture.hook.pendingId); + expect(r.guard(r.screen, '', r.epoch)).toBe('grant'); + expect(r.guard(r.screen, '', r.epoch)).toBe('handled'); +}); + +test('harmless preceding line number relocation keeps the exact content binding', () => { + const r = replay(f => { + f.before = 'Added unchanged line\n' + f.before; + f.screen = f.screen.replace(/^ {0,3}82 +$/m, ' 83 '); + }); + expect(r.epoch?.pendingId).toBe(fixture.hook.pendingId); +}); + +const negatives: Array<[string, (f: any) => void]> = [ + ['wrong preceding line', f => { f.screen = f.screen.replace(/^ {0,3}82 +$/m, ' 83 '); }], + ['unrelated prose continuation', f => { f.screen = f.screen.replace(/^.*\n/, ' Apply this edit now\n'); }], + ['source quotation continuation', f => { f.screen = f.screen.replace(/^.*\n/, ' > Example: apply this edit now\n'); }], + ['foreign continuation suffix', f => { f.screen = f.screen.replace('f2 (~7.6:1)', 'foreign (~7.6:1)'); }], + ['five-space unnumbered gutter', f => { f.screen = f.screen.slice(1); }], + ['seven-space unnumbered gutter', f => { f.screen = ' ' + f.screen; }], + ['no following unchanged numbered row', f => { f.screen = f.screen.replace(/^ {0,3}82 +$/m, ' 82 +'); }], + ['two arbitrary continuation rows', f => { f.screen = f.screen.split('\n')[0] + '\n' + f.screen; }], + ['missing original file', f => { f.fileKind = 'missing'; }], + ['directory in place of original file', f => { f.fileKind = 'directory'; }], + ['required original line beyond the bounded prefix', f => { f.before = 'x'.repeat(65537) + f.before; }], + ['foreign displayed directory', f => { f.screen = f.screen.replace(path.dirname(f.expected) + ' for this session', path.join(path.dirname(f.expected), 'foreign') + ' for this session'); }], + ['foreign hook expected path', f => { f.state.expected += '.foreign'; }], + ['no current native request', f => { f.state.pendingId = null; }], + ['completed native request', f => { f.state.completedId = f.state.pendingId; }], + ['stale native timestamp', f => { f.state.timestamp = new Date(f.startedAt - 1).toISOString(); }], + ['future native timestamp', f => { f.state.timestamp = new Date(Date.now() + 60000).toISOString(); }], + ['foreign session transcript', f => { f.transcript.assistantMessages[0].sessionId = 'foreign'; }], + ['unavailable native transcript', f => { f.transcript.status = 'error'; }], + ['missing complete menu footer', f => { f.screen = f.screen.replace('Esc to cancel · Tab to amend', ''); }], + ['selected policy change', f => { f.screen = f.screen.replace('❯ 1. Yes', '❯ 1. Yes, always allow'); }], + ['unrelated prompt prepended', f => { f.screen = '☐ Review this example\n' + f.screen; }], + ['historical pane is not current viewport', f => { f.screen = 'Waiting for current tool'; }], +]; +test.each(negatives)('%s cannot supply an owned epoch', (_, change) => { + const r = replay(change); expect(r.epoch).not.toBeTruthy(); expect(r.run()).not.toBe('grant'); +}); +test.skipIf(process.platform === 'win32')('a symlink cannot supply current original content', () => { + expect(replay(f => { f.fileKind = 'symlink'; }).epoch).toBeNull(); +}); +}); + +describe('plan-count-permission-ac', () => { +const captured = capturedAc; +const capturedAd = capturedAd_plan_count_permission_ac; +const capturedAe = capturedAe_plan_count_permission_ac; +const capturedAh = capturedAh_plan_count_permission_ac; +function fixture() { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'count-permission-ac-')); + const cwd = path.join(dir, 'cwd'), config = path.join(dir, '.claude'); + const expected = path.join(dir, 'report.md'), file = path.join(dir, 'state.json'); + fs.mkdirSync(cwd); const startedAt = Date.now() - 1000; + const screen = captured.rows[0]!.screen + .replace('../gstack-e2e-plan-design-3vPM9g/gstack-test-plan-design.md', expected) + .replaceAll('gstack-test-plan-design.md', 'report.md'); + const transcript: any = { status: 'ready', calls: [], assistantMessages: [{ sessionId: 'main', text: 'Reviewing' }] }; + const record = (name: string, id: string, delta: object = {}) => recordFilePermission(JSON.stringify({ + hook_event_name: name, tool_name: 'Edit', session_id: 'main', tool_use_id: id, cwd, + transcript_path: path.join(config, 'projects', 'owned', 'main.jsonl'), + tool_input: { file_path: expected, old_string: 'PRIVATE_OLD', new_string: 'PRIVATE_NEW' }, ...delta, + }), file, cwd, config, expected); + const epoch = () => currentFilePermissionEpoch(file, expected, cwd, config, startedAt, transcript, screen); + return { dir, cwd, config, expected, file, screen, record, epoch, + close: () => fs.rmSync(dir, { recursive: true, force: true }) }; +} + +test('all five actual stalled screens already identify an edit permission', () => { + for (const row of captured.rows) { + expect(classifyPlanCountFrame(row.screen), row.attempt).toBe('permission'); + expect(createPlanCountPermissionGuard()(row.screen), row.attempt).toBe('grant'); + expect(row.hook.pendingId).not.toBeNull(); + expect(row.transcriptSessions).toEqual([row.hook.sessionId]); + } +}); + +test('an intervening successful edit does not erase the exact previous grant completion', () => { + const f = fixture(); try { + const guard = createPlanCountPermissionGuard(), input = () => guard(f.screen, '', f.epoch()); + f.record('PreToolUse', 'granted'); expect(input()).toBe('grant'); + f.record('PostToolUse', 'granted'); expect(input()).toBe('handled'); + f.record('PreToolUse', 'automatic'); f.record('PostToolUse', 'automatic'); + // Old pane is still inert, even after two successful results. + expect(input()).toBe('handled'); + f.record('PreToolUse', 'next'); expect(input()).toBe('grant'); expect(input()).toBe('handled'); + expect(fs.readFileSync(f.file, 'utf8')).not.toContain('PRIVATE_'); + } finally { f.close(); } +}); + +test('an unrelated success cannot substitute for failed or missing prior approval completion', () => { + for (const outcome of ['PostToolUseFailure', 'missing', 'foreign', 'other-path', 'sidechain']) { + const f = fixture(); try { + const guard = createPlanCountPermissionGuard(), input = () => guard(f.screen, '', f.epoch()); + f.record('PreToolUse', 'granted'); expect(input()).toBe('grant'); + if (outcome === 'PostToolUseFailure') f.record(outcome, 'granted'); + else if (outcome !== 'missing') f.record('PostToolUse', 'granted', outcome === 'foreign' + ? { session_id: 'foreign' } : outcome === 'sidechain' ? { agent_id: 'child' } + : { tool_input: { file_path: path.join(f.dir, 'other.md') } }); + f.record('PreToolUse', 'automatic'); f.record('PostToolUse', 'automatic'); + f.record('PreToolUse', 'next'); expect(input(), outcome).toBe('handled'); + f.record('PreToolUse', 'granted'); expect(input(), outcome).toBe('handled'); + } finally { f.close(); } + } +}); + +test('success history rejects malformed, foreign, replayed, and pending IDs', () => { + const f = fixture(); try { + f.record('PreToolUse', 'first'); f.record('PostToolUse', 'first'); f.record('PreToolUse', 'next'); + const original = JSON.parse(fs.readFileSync(f.file, 'utf8')); + for (const completedIds of [['foreign:first'], ['main:../escape'], ['main:next'], + ['main:first', 'main:first'], Array(129).fill('main:first'), ['main:unseen'], 'main:first']) { + fs.writeFileSync(f.file, JSON.stringify({ ...original, completedIds })); + expect(f.epoch()).toBeNull(); + } + } finally { f.close(); } +}); + +test('success history is bounded by the existing 128-request recorder limit', () => { + const f = fixture(); try { + for (let i = 0; i < 127; i++) { f.record('PreToolUse', `id${i}`); f.record('PostToolUse', `id${i}`); } + f.record('PreToolUse', 'last'); expect(f.epoch()?.completedIds?.length).toBe(127); + expect(fs.statSync(f.file).size).toBeLessThan(64 * 1024); + f.record('PreToolUse', 'overflow'); expect(f.epoch()).toBeNull(); + } finally { f.close(); } +}); + +test('cropped actual panes bind their full directory and basename to the current native epoch', () => { + const rows = captured.rows.filter(row => [4, 5].includes(row.job) && !/^ {0,3}Edit file$/m.test(row.screen)); + expect(rows).toHaveLength(2); + for (const row of rows) { + const f = fixture(); try { + const screen = row.screen.replaceAll(path.dirname(row.hook.expected), path.dirname(f.expected)) + .replaceAll(path.basename(row.hook.expected), path.basename(f.expected)); + const read = (value = screen) => currentFilePermissionEpoch(f.file, f.expected, f.cwd, f.config, + 0, { status: 'ready', calls: [], assistantMessages: [{ sessionId: 'main', text: 'Reviewing', timestamp: new Date().toISOString() }] }, value); + f.record('PreToolUse', 'current'); + expect(read()?.pendingId, row.attempt).toBe('main:current'); + const guard = createPlanCountPermissionGuard(); + expect(guard(screen, '', read())).toBe('grant'); + expect(guard(screen, '', read())).toBe('handled'); + for (const [name, changed] of [ + ['foreign', screen.replace(path.dirname(f.expected), path.join(f.dir, 'foreign'))], + ['remedy', screen.replace(/always\s+allow\s+access\s+to/, 'remove files from')], + ['footer', screen.replace('Esc to cancel · Tab to amend', '')], + ['yes policy', screen.replace(/❯\s*1\.\s*Yes/, '❯ 1. Yes, change policy')], + ['AUQ', '☐ Finding\n' + screen], + ['quoted', '> Example:\n' + screen], + ['wrapped path', screen.replace(path.dirname(f.expected), path.dirname(f.expected) + '\n/other')], + ['path spaces', screen.replace(path.dirname(f.expected), path.dirname(f.expected) + ' space')], + ]) { + expect(changed, name).not.toBe(screen); + expect(read(changed), name).toBeNull(); + } + expect(read(screen.replace(path.dirname(f.expected), path.join(f.dir, 'foreign')))).toBeNull(); + f.record('PostToolUse', 'current'); expect(read()).toBeNull(); + f.record('PreToolUse', 'current'); expect(read()).toBeNull(); + } finally { f.close(); } + } +}); +test('a later exact owned binding wins over an earlier same-basename block', () => { + const f = fixture(); try { + f.record('PreToolUse', 'current'); + const transcript: any = {status:'ready', calls:[], assistantMessages:[{sessionId:'main', text:'Reviewing'}]}; + const foreign = {file:path.join(f.dir,'foreign-state.json'), expected:path.join(f.dir,'other','report.md')}; + const owned = {file:f.file, expected:f.expected}; + for (const bindings of [[foreign, owned], [owned, foreign]]) { + const selected = currentFilePermissionBinding(bindings, f.cwd, f.config, 0, transcript, f.screen); + expect(selected?.binding).toBe(owned); + expect(selected?.epoch.pendingId).toBe('main:current'); + } + const blocked = currentFilePermissionBinding([foreign, {...foreign, expected:path.join(f.dir,'another','report.md')}], + f.cwd, f.config, 0, transcript, f.screen); + expect(blocked).toBeNull(); + expect(createPlanCountPermissionGuard()(f.screen, '', blocked)).toBe('handled'); + const otherScreen = f.screen.replaceAll('report.md', 'OTHER.md'); + const unrelated = currentFilePermissionBinding([foreign, owned], f.cwd, f.config, 0, transcript, otherScreen); + expect(unrelated).toBeUndefined(); + expect(createPlanCountPermissionGuard()(otherScreen, '', unrelated)).toBe('grant'); + } finally { f.close(); } +}); + +// Exact current screens plus content-free native identity from full AD/AE runs. +// The replay projections do not assert these pending writes ever completed. +const cases = [...capturedAd.rows, capturedAe, capturedAh].map(row => ({ + p: row, screen: row.screen, binding: {expected: row.state.expected, state: row.state}, + observation: {transcript: {status: row.transcriptStatus, calls: [], + assistantMessages: row.transcriptSessions.map(sessionId => ({sessionId}))}}, +})); +function adEpoch(c: any, screen = c.screen, state = c.binding.state, delta: any = {}) { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'permission-crop-replay-')); + const file = path.join(dir, 'state.json'); + try { + // Captures contain POSIX paths. Project their filesystem identity onto the + // replay host without changing the captured fixture or its menu rendering. + const nativeState = { ...state, cwd: path.resolve(state.cwd), expected: path.resolve(state.expected), + transcriptPath: path.resolve(state.transcriptPath) }; + const replayScreen = screen.replaceAll(path.posix.dirname(c.binding.expected), + path.dirname(path.resolve(c.binding.expected))); + fs.writeFileSync(file, JSON.stringify(nativeState)); + return currentFilePermissionEpoch(file, path.resolve(delta.expected ?? c.binding.expected), path.resolve(delta.cwd ?? c.p.cwd), + path.resolve(delta.config ?? c.p.config), delta.startedAt ?? c.p.startUnix * 1000, + delta.transcript ?? c.observation.transcript, replayScreen); + } finally { fs.rmSync(dir, {recursive:true, force:true}); } +} +for (const c of cases) { + test(`actual AD captured current permission ${c.p.pid} returns its exact pending epoch`, () => { + expect(classifyPlanCountFrame(c.screen)).toBe('permission'); + expect(adEpoch(c)).toEqual({pendingId:c.binding.state.pendingId, completedId:c.binding.state.completedId, + completedIds:c.binding.state.completedIds}); + const guard = createPlanCountPermissionGuard(); + expect(guard(c.screen, '', adEpoch(c))).toBe('grant'); + expect(guard(c.screen, '', adEpoch(c))).toBe('handled'); + }); + test(`AD isolating the rejected rendering guard ${c.p.pid} preserves native identity`, () => { + // These are explicitly normalized controls; the actual captured screen is unchanged above. + const normalized = c.p.pid === capturedAe.pid ? c.screen.replace(/^[╌─━]{3,}[ \t]*\n/, '') + : c.p.pid === 1332470 ? c.screen.replace('3. Nohift+tab)', '3. No') : c.screen.replace(/^ {4,5}\+/, ' 99 +'); + expect(normalized).not.toBe(c.screen); + expect(adEpoch(c, normalized)?.pendingId).toBe(c.binding.state.pendingId); + }); +} +for (const c of cases) { + test(`AD crop ${c.p.pid} rejects foreign, quoted, incomplete and policy-changing menus`, () => { + const directory = path.dirname(c.binding.expected); + const changes: [string, string][] = [ + ['foreign directory', c.screen.replace(directory, path.join(directory, 'foreign'))], + ['split directory', c.screen.replace(directory, directory + '\n/foreign')], + ['quoted', '> Example:\n' + c.screen], + ['AUQ', '☐ Review\n' + c.screen], + ['code fence', '```\n' + c.screen], + ['missing footer', c.screen.replace('Esc to cancel · Tab to amend', '')], + ['policy on selected Yes', c.screen.replace('❯ 1. Yes', '❯ 1. Yes, always allow')], + ['selected No', c.screen.replace('❯ 1. Yes', ' 1. Yes').replace(' 3. No', ' ❯ 3. No')], + ['arbitrary No suffix', c.screen.replace(/3\. No(?:hift\+tab\))?/, '3. No; run another command')], + ['another hint', c.screen.replace(/3\. No(?:hift\+tab\))?/, '3. No(shift+enter)')], + ['foreign option action', c.screen.replace(/always\s+allow\s+access\s+to/, 'delete files from')], + ['unrecognized cropped prose', c.screen.replace(/^.*\n/, 'arbitrary text\n')], + ]; + for (const [name, screen] of changes) { + expect(screen, name).not.toBe(c.screen); + expect(adEpoch(c, screen), name).not.toBeTruthy(); + expect(createPlanCountPermissionGuard()(screen, '', adEpoch(c, screen)), name).not.toBe('grant'); + } + }); + test(`AD crop ${c.p.pid} leaves unrelated-basename permission policy unchanged`, () => { + const screen = c.screen.replace(path.basename(c.binding.expected), 'OTHER.md'); + expect(adEpoch(c, screen)).toBeUndefined(); + // This is intentionally the existing caller policy for unrelated fixture permissions. + expect(createPlanCountPermissionGuard()(screen, '', adEpoch(c, screen))).toBe('grant'); + }); + test(`AD crop ${c.p.pid} cannot replace missing, stale, completed or foreign native identity`, () => { + const original = c.binding.state; + for (const [name, state, delta] of [ + ['foreign session', {...original, sessionId:'foreign'}, {}], + ['wrong native transcript', {...original, transcriptPath:path.join(c.p.config, 'projects', 'foreign', 'other.jsonl')}, {}], + ['completed request', {...original, completedId:original.pendingId}, {}], + ['no pending request', {...original, pendingId:null}, {}], + ['unseen pending request', {...original, pendingId:original.sessionId + ':other'}, {}], + ['stale timestamp', {...original, timestamp:new Date(c.p.startUnix * 1000 - 1).toISOString()}, {}], + ['future timestamp', {...original, timestamp:new Date(Date.now() + 60_000).toISOString()}, {}], + ['mixed sessions', original, {transcript:{status:'ready', calls:[], assistantMessages:[{sessionId:original.sessionId},{sessionId:'foreign'}]}}], + ['unready transcript', original, {transcript:{...c.observation.transcript,status:'unavailable'}}], + ] as const) { + expect(adEpoch(c, c.screen, state, delta), name).toBeNull(); + } + }); +} + +test('AD crop fixture selects the exact existing permission regression callers', () => { + expect(selectTests(['test/fixtures/plan-count-permission-ad.json'], E2E_TOUCHFILES).selected.sort()).toEqual( + selectTests(['test/fixtures/plan-count-permission-ac.json'], E2E_TOUCHFILES).selected.sort()); +}); + +test('AE crop admits one native divider only and preserves its exact existing caller selection', () => { + const c = cases.find(item => item.p.pid === capturedAe.pid)!; + const firstLine = c.screen.slice(0, c.screen.indexOf('\n') + 1); + for (const screen of [firstLine + c.screen, 'unrelated prose\n' + c.screen, + firstLine + 'Example:\n' + c.screen.slice(firstLine.length), + c.screen.replace(firstLine, firstLine.trimEnd() + ' extra action\n')]) { + expect(adEpoch(c, screen)).toBeNull(); + expect(createPlanCountPermissionGuard()(screen, '', adEpoch(c, screen))).not.toBe('grant'); + } + expect(selectTests(['test/fixtures/plan-count-permission-ae.json'], E2E_TOUCHFILES).selected.sort()).toEqual( + selectTests(['test/fixtures/plan-count-permission-ad.json'], E2E_TOUCHFILES).selected.sort()); +}); + +test('AH wrapped crop admits four or five spaces with the same owned native epoch', () => { + const c = cases.find(item => item.p.pid === capturedAh.pid)!; + expect(c.screen.startsWith(' + ')).toBe(true); + for (const screen of [c.screen, ` ${c.screen}`, c.screen.replace(/^ \+/, ' -')]) { + expect(adEpoch(c, screen)?.pendingId).toBe(c.binding.state.pendingId); + const guard = createPlanCountPermissionGuard(); + expect(guard(screen, '', adEpoch(c, screen))).toBe('grant'); + expect(guard(screen, '', adEpoch(c, screen))).toBe('handled'); + } + // Cleaning the unselected No paint residue does not establish missing identity. + const noOnly = c.screen.replace('3. Nohift+tab)', '3. No'); + expect(noOnly).not.toBe(c.screen); + expect(adEpoch(c, noOnly)?.pendingId).toBe(c.binding.state.pendingId); +}); + +test('AH continuation crop rejects prose, unsupported gutters and malformed numbered context', () => { + const c = cases.find(item => item.p.pid === capturedAh.pid)!; + for (const [name, screen] of [ + ['four-space prose', c.screen.replace(/^.*\n/, ' Apply this edit now\n')], + ['four-space quoted prose', c.screen.replace(/^.*\n/, ' > Example\n')], + ['three-space gutter', c.screen.slice(1)], + ['six-space gutter', ` ${c.screen}`], + ['no numbered rows', c.screen.replace(/^\s*\d+\s+(?=[+\- ])/gm, ' +')], + ['one numbered row', c.screen.replace(/^(\s*\d+\s+)(?=[+\- ])/gm, + (prefix, _group, offset) => offset === c.screen.indexOf(' 79 ') ? prefix : ' +')], + ]) { + expect(screen, name).not.toBe(c.screen); + expect(adEpoch(c, screen), name).toBeNull(); + expect(createPlanCountPermissionGuard()(screen, '', adEpoch(c, screen)), name).not.toBe('grant'); + } + expect(selectTests(['test/fixtures/plan-count-permission-ah.json'], E2E_TOUCHFILES).selected.sort()).toEqual( + selectTests(['test/fixtures/plan-count-permission-ae.json'], E2E_TOUCHFILES).selected.sort()); +}); +}); + +describe('plan-count-quoted-frame-ak', () => { +const fs = fs_batching_permission_at; +const os = os_batching_permission_at; +const path = path_batching_permission_at; +const exact = exact_plan_count_quoted_frame_ak; +const native = capturedAc; +const owned = owned_plan_count_quoted_frame_ak; +const quote=(s:string)=>s.split('\n').map(row=>'> '+row).join('\n'); + +// A PTY transports bytes, not command-sized stdin events. Share the framing +// code with the fake CLI so fragmented grants exercise the same receiver. +function commandBuffer(){ + let pending=''; + return (chunk:string)=>{ + pending+=chunk;const commands:string[]=[];let end:number; + while((end=pending.indexOf('\r'))!==-1){commands.push(pending.slice(0,end+1));pending=pending.slice(end+1);} + return commands; + }; +} +test('fake CLI preserves command bytes across fragmented and coalesced PTY input',()=>{ + const expected=['/plan-ceo-review\r','1\r']; + for(const chunks of [expected,['/plan-ceo-review\r','1','\r'],[...expected.join('')],[expected.join('')]]){ + const receive=commandBuffer();expect(chunks.flatMap(receive)).toEqual(expected); + } + const receive=commandBuffer(); + expect(receive('1')).toEqual([]);expect(receive('\r2\rtrailing')).toEqual(['1\r','2\r']); + expect(receive('\r')).toEqual(['trailing\r']); // no unexpected bytes are discarded +}); + +test('the exact wholly quoted AK pane is handled so the dispatcher sends no fallback',()=>{ + const screen=quote(exact.screen); + expect(classifyPlanCountFrame(screen)).toBe('permission'); + expect(createPlanCountPermissionGuard()(screen,'',undefined)).toBe('handled'); +}); +test('quoted whole native frames remain inert across redraw, indentation, CRLF and terminal color',()=>{ + for(const source of [exact.screen,native.rows[0]!.screen,owned.screen])for(const transform of [ + (s:string)=>quote(s),(s:string)=>'\n'+quote(s)+'\n',(s:string)=>quote(s).replace(/^>/gm,' >'), + (s:string)=>quote(s).replaceAll('\n','\r\n'),(s:string)=>'\x1b[31m'+quote(s)+'\x1b[0m', + ]){const s=transform(source),g=createPlanCountPermissionGuard(); + const expected=classifyPlanCountFrame(s)==='permission'?'handled':null; + expect(g(s,'')).toBe(expected);expect(g(s,'⎿ Wrote 4 lines\n')).toBe(expected);} +}); +test('quoted history cannot consume or authorize the current native grant',()=>{ + const s=native.rows[0]!.screen,g=createPlanCountPermissionGuard(),q=quote(s); + expect(g(q,'')).toBe('handled');expect(g(s,q)).toBe('grant');expect(g(q,s)).toBe('handled');expect(g(s,q)).toBe('handled'); + expect(createPlanCountPermissionGuard()(s,'',null)).toBe('handled'); + expect(createPlanCountPermissionGuard()(s,'',{pendingId:'main:first',completedId:null})).toBe('grant'); + expect(createPlanCountPermissionGuard()(q,'',{pendingId:'main:first',completedId:null})).toBe('handled'); +}); +test('unquoted current native frames retain their policy even with quote characters in diff text or history',()=>{ + for(const s of native.rows.map(row=>row.screen))expect(createPlanCountPermissionGuard()(s,quote(s))).toBe('grant'); + expect(createPlanCountPermissionGuard()('> An old note\n'+native.rows[0]!.screen,'')).toBe('grant'); + expect(createPlanCountPermissionGuard()('No current pane',quote(exact.screen))).toBeNull(); + expect(createPlanCountPermissionGuard()('> Ordinary quoted prose, without a permission menu')).toBeNull(); +}); +test.skipIf(process.platform==='win32')('real dispatcher ignores quoted pane then grants the fresh owned native request once',async()=>{ + const dir=fs.mkdtempSync(path.join(os.tmpdir(),'count-quoted-frame-')),fake=path.join(dir,'fake-claude'),worker=path.join(dir,'worker.ts'),events=path.join(dir,'events.jsonl'),output=path.join(dir,'result.json'),report=path.join(dir,'report.md'); + fs.writeFileSync(report,'original'); + fs.writeFileSync(fake,`#!${process.execPath}\nconst receive=(${commandBuffer.toString()})();\n`+String.raw` +import fs from 'node:fs';import path from 'node:path'; +const item=JSON.parse(process.env.QUOTED_FRAME_CASE),sid='quoted-frame-main',log=e=>fs.appendFileSync(item.events,JSON.stringify(e)+'\n'); +const transcript=path.join(process.env.CLAUDE_CONFIG_DIR,'projects','owned',sid+'.jsonl');fs.mkdirSync(path.dirname(transcript),{recursive:true}); +const native=(role,content,extra={})=>fs.appendFileSync(transcript,JSON.stringify({cwd:process.cwd(),sessionId:sid,isSidechain:false,timestamp:new Date().toISOString(),message:{role,content},...extra})+'\n'); +native('assistant',[{type:'text',text:'Reviewing the owned fixture.'}]);log({type:'start',pid:process.pid,cwd:process.cwd()}); +const settings=JSON.parse(process.argv[process.argv.indexOf('--settings')+1]); +const hook=async name=>{for(const entry of settings.hooks[name]??[]){if(entry.matcher!=='^(Write|Edit)$')continue; + const event={hook_event_name:name,tool_name:'Edit',session_id:sid,tool_use_id:'current',cwd:process.cwd(),transcript_path:transcript,tool_input:{file_path:item.report}}; + const p=Bun.spawn(['bash','-c',entry.hooks[0].command],{stdin:new Blob([JSON.stringify(event)]),stdout:'pipe',stderr:'pipe'}); + const [code,out,err]=await Promise.all([p.exited,new Response(p.stdout).text(),new Response(p.stderr).text()]);if(code||out||err)throw Error('Hook failed');}}; +const pane=item.screen.replaceAll('PLAN.md',item.report),paint=s=>process.stdout.write('\x1b[2J\x1b[H'+s.replaceAll('\n','\r\n')); +let stage='startup';process.stdin.setRawMode?.(true);const dispatch=async input=>{ + log({type:'input',stage,input}); + if(stage==='startup'){stage='quoted';paint(item.quotedScreen);setTimeout(async()=>{await hook('PreToolUse');stage='current';paint(pane);},4200);return;} + if(stage!=='current'){log({type:'unexpected'});return;} + if(input!=='1\r')throw Error('One-time grant changed');stage='done';await hook('PostToolUse'); + const q={header:'Finding',question:'Apply the reviewed fix?',options:[{label:'Fix'},{label:'Keep'}]}; + native('assistant',[{type:'tool_use',name:'AskUserQuestion',id:'finding',input:{questions:[q]}}]);native('user',[{type:'tool_result',tool_use_id:'finding',content:'Answered'}],{toolUseResult:{answers:{[q.question]:'Fix'}}});paint('Done.\n'); +};process.stdin.on('data',async data=>{const chunk=data.toString();log({type:'chunk',stage,input:chunk});for(const input of receive(chunk))await dispatch(input);});process.on('SIGINT',()=>process.exit(0));process.stdin.resume(); +process.stdout.write('PTY_READY:'+item.events+'\x1b[2J\x1b[H'); +`);fs.chmodSync(fake,0o755); + // Keep every physical terminal row inside the quote; adding a prefix to an + // already120-column capture would otherwise wrap an unquoted continuation. + const quotedScreen=exact.screen.split('\n').flatMap(row=>row.trimEnd().match(/.{1,116}/gu)??['']).map(row=>'> '+row).join('\n'); + const args={skillName:'plan-ceo-review',slashCommand:'/plan-ceo-review',followUpPrompt:'Review the disposable fixture.',expectedPlanPath:report,reviewCountCeiling:1,timeoutMs:25000,startupReadyMarker:'PTY_READY:'+events,env:{QUOTED_FRAME_CASE:JSON.stringify({events,report,screen:owned.screen,quotedScreen})}}; + fs.writeFileSync(worker,`import {runPlanSkillCounting} from ${JSON.stringify(pathToFileURL(path.join(import.meta.dir,'helpers/claude-pty-runner.ts')).href)};const result=await runPlanSkillCounting({...${JSON.stringify(args)},isLastStep0AUQ:()=>false,isReviewAUQ:()=>true});await Bun.write(${JSON.stringify(output)},JSON.stringify(result));`); + const child=Bun.spawn([process.execPath,worker],{env:{...process.env,BROWSE_TERMINAL_BINARY:fake,EVALS_HERMETIC:'1'},stdout:'pipe',stderr:'pipe'}),timer=setTimeout(()=>child.kill('SIGKILL'),30000); + try{const [code,out,err]=await Promise.all([child.exited,new Response(child.stdout).text(),new Response(child.stderr).text()]);expect(code,out+err).toBe(0); + const result=JSON.parse(fs.readFileSync(output,'utf8')),rows=fs.readFileSync(events,'utf8').trim().split('\n').map(s=>JSON.parse(s)); + expect(result.outcome,JSON.stringify({result,rows})).toBe('ceiling_reached');expect(result.reviewCount).toBe(1); + const chunks=rows.filter(r=>r.type==='chunk'); + expect(chunks.every(r=>['startup','current'].includes(r.stage))).toBe(true); + expect(chunks.map(r=>r.input).join('')).toBe('/plan-ceo-review\r1\r'); + expect(rows.filter(r=>r.type==='input').map(r=>[r.stage,r.input])).toEqual([['startup','/plan-ceo-review\r'],['current','1\r']]);expect(rows.some(r=>r.type==='unexpected')).toBe(false); + expect(()=>process.kill(rows[0].pid,0)).toThrow();expect(fs.existsSync(rows[0].cwd)).toBe(false); + }finally{clearTimeout(timer);child.kill('SIGKILL');await child.exited; + if(fs.existsSync(events)){const first=JSON.parse(fs.readFileSync(events,'utf8').split('\n')[0]!);try{if(fs.readFileSync('/proc/'+first.pid+'/cmdline','utf8').split('\0').includes(fake))process.kill(first.pid,'SIGKILL');}catch{}} + fs.rmSync(dir,{recursive:true,force:true});} +},32000); +}); diff --git a/test/plan-count-navigation-r.test.ts b/test/plan-count-navigation-r.test.ts deleted file mode 100644 index c9aed08e1..000000000 --- a/test/plan-count-navigation-r.test.ts +++ /dev/null @@ -1,79 +0,0 @@ -import { describe, expect, test } from 'bun:test'; -import { capturePlanCountQuestion, nativePlanCallFingerprint, planCountPrerequisitePick, planCountQuestionInput } from './helpers/claude-pty-runner'; -import type { NativePlanQuestionCall } from './helpers/plan-count-transcript'; -import prerequisite from './fixtures/dx-prerequisite-r-call.json'; - -function pending(source: NativePlanQuestionCall): NativePlanQuestionCall { - const call = structuredClone(source); - call.answered = false; - delete call.answers; - delete call.unansweredQuestionIndices; - delete call.answeredAt; - return call; -} -function frame(call: NativePlanQuestionCall) { - const q = call.questions[0]!; - const visible = `☐ ${q.header}\n${q.question}\n${q.options.map((o, i) => `${i ? ' ' : '❯'} ${i + 1}. ${o.label}`).join('\n')}\nEnter to select · ↑/↓ to navigate · Esc to cancel`; - const active = capturePlanCountQuestion(visible, new Set(), 0, true, call)!; - return { visible, active, routing: nativePlanCallFingerprint(call, 0, true) }; -} - -describe('captured R planning navigation', () => { - test('DX declines its optional office-hours detour using the full native option meaning', () => { - const call = pending(prerequisite as NativePlanQuestionCall); - const { visible, active, routing } = frame(call); - expect(active.nativeCall).toBe(call); - expect(prerequisite.answers[prerequisite.questions[0]!.question]).toBe(prerequisite.questions[0]!.options[0]!.label); - const pick = planCountPrerequisitePick(routing, active) ?? 1; - expect(pick).toBe(2); - expect(planCountQuestionInput(visible, active, pick)).toBe('2'); - }); - test('short prerequisite labels require the current native question and affirmative review action', () => { - for (const change of [ - (c: NativePlanQuestionCall) => { c.questions[0]!.options[1]!.description = ''; }, - (c: NativePlanQuestionCall) => { c.questions[0]!.options[1]!.description = 'Do not proceed with standard DX POLISH review.'; }, - (c: NativePlanQuestionCall) => { c.questions[0]!.options[1]!.description = 'Accept the finding and proceed with standard DX POLISH review.'; }, - (c: NativePlanQuestionCall) => { c.questions[0]!.options[1]!.description = 'Accept this security finding. Proceed with standard DX POLISH review.'; }, - (c: NativePlanQuestionCall) => { c.questions[0]!.options[1]!.description = 'Proceed with standard DX POLISH review after running /office-hours first.'; }, - (c: NativePlanQuestionCall) => { c.questions[0]!.options[1]!.description = 'Proceed with standard DX POLISH review? No, run /office-hours first.'; }, - (c: NativePlanQuestionCall) => { c.questions[0]!.options[1]!.description = 'Proceed with standard DX POLISH review. Accept the security risk.'; }, - (c: NativePlanQuestionCall) => { c.questions[0]!.options[1]!.description = 'Plan scope is already precise. Proceed with standard DX POLISH review if the tests pass.'; }, - (c: NativePlanQuestionCall) => { c.questions[0]!.options.push({ label: 'Accept this security finding' }); }, - (c: NativePlanQuestionCall) => { c.questions[0]!.multiSelect = true; }, - (c: NativePlanQuestionCall) => { c.questions.push(structuredClone(c.questions[0]!)); }, - (c: NativePlanQuestionCall) => { c.answered = true; }, - (c: NativePlanQuestionCall) => { c.failed = true; }, - ]) { - const call = pending(prerequisite as NativePlanQuestionCall); - change(call); - const fp = nativePlanCallFingerprint(call, 0, true); - expect(planCountPrerequisitePick(fp)).toBeNull(); - } - const call = pending(prerequisite as NativePlanQuestionCall); - const { active, routing, visible } = frame(call); - expect(planCountPrerequisitePick(routing, { ...active, preReview: false })).toBeNull(); - expect(planCountPrerequisitePick(routing, { ...active, signature: 'stale:question' })).toBeNull(); - expect(planCountPrerequisitePick(routing, { ...active, nativeQuestionIndex: 1 })).toBeNull(); - expect(planCountPrerequisitePick(routing, { ...active, options: active.options.slice().reverse() })).toBeNull(); - const visibleOnly = capturePlanCountQuestion(visible, new Set(), 0, true)!; - expect(visibleOnly.nativeCall).toBeUndefined(); - expect(planCountPrerequisitePick(routing, visibleOnly)).toBeNull(); - call.questions[0]!.options.reverse(); - const reordered = frame(call); - expect(planCountPrerequisitePick(reordered.routing, reordered.active)).toBe(1); - }); - - test('a bare short label uses its bound unconditional review meaning', () => { - const call = pending(prerequisite as NativePlanQuestionCall);call.questions[0]!.options[1]!.label='Skip'; - const {active,routing}=frame(call);expect(planCountPrerequisitePick(routing,active)).toBe(2); - }); - - test('short labels allow only a benign plan-scope premise plus the unconditional review action', () => { - for (const description of ['Proceed with standard review.', 'Proceed with standard DX review', 'The plan is clear. Proceed with standard DX POLISH review.', 'Plan scope is already precise. Proceed with standard DX POLISH review.']) { - const call = pending(prerequisite as NativePlanQuestionCall); - call.questions[0]!.options[1]!.description = description; - const { active, routing } = frame(call); - expect(planCountPrerequisitePick(routing, active)).toBe(2); - } - }); -}); diff --git a/test/plan-count-permission-ac.test.ts b/test/plan-count-permission-ac.test.ts deleted file mode 100644 index 9c8a7fa2c..000000000 --- a/test/plan-count-permission-ac.test.ts +++ /dev/null @@ -1,283 +0,0 @@ -import { expect, test } from 'bun:test'; -import * as fs from 'node:fs'; -import * as os from 'node:os'; -import * as path from 'node:path'; -import captured from './fixtures/plan-count-permission-ac.json'; -import capturedAd from './fixtures/plan-count-permission-ad.json'; -import capturedAe from './fixtures/plan-count-permission-ae.json'; -import capturedAh from './fixtures/plan-count-permission-ah.json'; -import { classifyPlanCountFrame, createPlanCountPermissionGuard } from './helpers/claude-pty-runner'; -import { recordFilePermission, currentFilePermissionEpoch, currentFilePermissionBinding } from './helpers/plan-count-file-permission'; -import { E2E_TOUCHFILES, selectTests } from './helpers/touchfiles'; - -function fixture() { - const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'count-permission-ac-')); - const cwd = path.join(dir, 'cwd'), config = path.join(dir, '.claude'); - const expected = path.join(dir, 'report.md'), file = path.join(dir, 'state.json'); - fs.mkdirSync(cwd); const startedAt = Date.now() - 1000; - const screen = captured.rows[0]!.screen - .replace('../gstack-e2e-plan-design-3vPM9g/gstack-test-plan-design.md', expected) - .replaceAll('gstack-test-plan-design.md', 'report.md'); - const transcript: any = { status: 'ready', calls: [], assistantMessages: [{ sessionId: 'main', text: 'Reviewing' }] }; - const record = (name: string, id: string, delta: object = {}) => recordFilePermission(JSON.stringify({ - hook_event_name: name, tool_name: 'Edit', session_id: 'main', tool_use_id: id, cwd, - transcript_path: path.join(config, 'projects', 'owned', 'main.jsonl'), - tool_input: { file_path: expected, old_string: 'PRIVATE_OLD', new_string: 'PRIVATE_NEW' }, ...delta, - }), file, cwd, config, expected); - const epoch = () => currentFilePermissionEpoch(file, expected, cwd, config, startedAt, transcript, screen); - return { dir, cwd, config, expected, file, screen, record, epoch, - close: () => fs.rmSync(dir, { recursive: true, force: true }) }; -} - -test('all five actual stalled screens already identify an edit permission', () => { - for (const row of captured.rows) { - expect(classifyPlanCountFrame(row.screen), row.attempt).toBe('permission'); - expect(createPlanCountPermissionGuard()(row.screen), row.attempt).toBe('grant'); - expect(row.hook.pendingId).not.toBeNull(); - expect(row.transcriptSessions).toEqual([row.hook.sessionId]); - } -}); - -test('an intervening successful edit does not erase the exact previous grant completion', () => { - const f = fixture(); try { - const guard = createPlanCountPermissionGuard(), input = () => guard(f.screen, '', f.epoch()); - f.record('PreToolUse', 'granted'); expect(input()).toBe('grant'); - f.record('PostToolUse', 'granted'); expect(input()).toBe('handled'); - f.record('PreToolUse', 'automatic'); f.record('PostToolUse', 'automatic'); - // Old pane is still inert, even after two successful results. - expect(input()).toBe('handled'); - f.record('PreToolUse', 'next'); expect(input()).toBe('grant'); expect(input()).toBe('handled'); - expect(fs.readFileSync(f.file, 'utf8')).not.toContain('PRIVATE_'); - } finally { f.close(); } -}); - -test('an unrelated success cannot substitute for failed or missing prior approval completion', () => { - for (const outcome of ['PostToolUseFailure', 'missing', 'foreign', 'other-path', 'sidechain']) { - const f = fixture(); try { - const guard = createPlanCountPermissionGuard(), input = () => guard(f.screen, '', f.epoch()); - f.record('PreToolUse', 'granted'); expect(input()).toBe('grant'); - if (outcome === 'PostToolUseFailure') f.record(outcome, 'granted'); - else if (outcome !== 'missing') f.record('PostToolUse', 'granted', outcome === 'foreign' - ? { session_id: 'foreign' } : outcome === 'sidechain' ? { agent_id: 'child' } - : { tool_input: { file_path: path.join(f.dir, 'other.md') } }); - f.record('PreToolUse', 'automatic'); f.record('PostToolUse', 'automatic'); - f.record('PreToolUse', 'next'); expect(input(), outcome).toBe('handled'); - f.record('PreToolUse', 'granted'); expect(input(), outcome).toBe('handled'); - } finally { f.close(); } - } -}); - -test('success history rejects malformed, foreign, replayed, and pending IDs', () => { - const f = fixture(); try { - f.record('PreToolUse', 'first'); f.record('PostToolUse', 'first'); f.record('PreToolUse', 'next'); - const original = JSON.parse(fs.readFileSync(f.file, 'utf8')); - for (const completedIds of [['foreign:first'], ['main:../escape'], ['main:next'], - ['main:first', 'main:first'], Array(129).fill('main:first'), ['main:unseen'], 'main:first']) { - fs.writeFileSync(f.file, JSON.stringify({ ...original, completedIds })); - expect(f.epoch()).toBeNull(); - } - } finally { f.close(); } -}); - -test('success history is bounded by the existing 128-request recorder limit', () => { - const f = fixture(); try { - for (let i = 0; i < 127; i++) { f.record('PreToolUse', `id${i}`); f.record('PostToolUse', `id${i}`); } - f.record('PreToolUse', 'last'); expect(f.epoch()?.completedIds?.length).toBe(127); - expect(fs.statSync(f.file).size).toBeLessThan(64 * 1024); - f.record('PreToolUse', 'overflow'); expect(f.epoch()).toBeNull(); - } finally { f.close(); } -}); - -test('cropped actual panes bind their full directory and basename to the current native epoch', () => { - const rows = captured.rows.filter(row => [4, 5].includes(row.job) && !/^ {0,3}Edit file$/m.test(row.screen)); - expect(rows).toHaveLength(2); - for (const row of rows) { - const f = fixture(); try { - const screen = row.screen.replaceAll(path.dirname(row.hook.expected), path.dirname(f.expected)) - .replaceAll(path.basename(row.hook.expected), path.basename(f.expected)); - const read = (value = screen) => currentFilePermissionEpoch(f.file, f.expected, f.cwd, f.config, - 0, { status: 'ready', calls: [], assistantMessages: [{ sessionId: 'main', text: 'Reviewing', timestamp: new Date().toISOString() }] }, value); - f.record('PreToolUse', 'current'); - expect(read()?.pendingId, row.attempt).toBe('main:current'); - const guard = createPlanCountPermissionGuard(); - expect(guard(screen, '', read())).toBe('grant'); - expect(guard(screen, '', read())).toBe('handled'); - for (const [name, changed] of [ - ['foreign', screen.replace(path.dirname(f.expected), path.join(f.dir, 'foreign'))], - ['remedy', screen.replace(/always\s+allow\s+access\s+to/, 'remove files from')], - ['footer', screen.replace('Esc to cancel · Tab to amend', '')], - ['yes policy', screen.replace(/❯\s*1\.\s*Yes/, '❯ 1. Yes, change policy')], - ['AUQ', '☐ Finding\n' + screen], - ['quoted', '> Example:\n' + screen], - ['wrapped path', screen.replace(path.dirname(f.expected), path.dirname(f.expected) + '\n/other')], - ['path spaces', screen.replace(path.dirname(f.expected), path.dirname(f.expected) + ' space')], - ]) { - expect(changed, name).not.toBe(screen); - expect(read(changed), name).toBeNull(); - } - expect(read(screen.replace(path.dirname(f.expected), path.join(f.dir, 'foreign')))).toBeNull(); - f.record('PostToolUse', 'current'); expect(read()).toBeNull(); - f.record('PreToolUse', 'current'); expect(read()).toBeNull(); - } finally { f.close(); } - } -}); -test('a later exact owned binding wins over an earlier same-basename block', () => { - const f = fixture(); try { - f.record('PreToolUse', 'current'); - const transcript: any = {status:'ready', calls:[], assistantMessages:[{sessionId:'main', text:'Reviewing'}]}; - const foreign = {file:path.join(f.dir,'foreign-state.json'), expected:path.join(f.dir,'other','report.md')}; - const owned = {file:f.file, expected:f.expected}; - for (const bindings of [[foreign, owned], [owned, foreign]]) { - const selected = currentFilePermissionBinding(bindings, f.cwd, f.config, 0, transcript, f.screen); - expect(selected?.binding).toBe(owned); - expect(selected?.epoch.pendingId).toBe('main:current'); - } - const blocked = currentFilePermissionBinding([foreign, {...foreign, expected:path.join(f.dir,'another','report.md')}], - f.cwd, f.config, 0, transcript, f.screen); - expect(blocked).toBeNull(); - expect(createPlanCountPermissionGuard()(f.screen, '', blocked)).toBe('handled'); - const otherScreen = f.screen.replaceAll('report.md', 'OTHER.md'); - const unrelated = currentFilePermissionBinding([foreign, owned], f.cwd, f.config, 0, transcript, otherScreen); - expect(unrelated).toBeUndefined(); - expect(createPlanCountPermissionGuard()(otherScreen, '', unrelated)).toBe('grant'); - } finally { f.close(); } -}); - -// Exact current screens plus content-free native identity from full AD/AE runs. -// The replay projections do not assert these pending writes ever completed. -const cases = [...capturedAd.rows, capturedAe, capturedAh].map(row => ({ - p: row, screen: row.screen, binding: {expected: row.state.expected, state: row.state}, - observation: {transcript: {status: row.transcriptStatus, calls: [], - assistantMessages: row.transcriptSessions.map(sessionId => ({sessionId}))}}, -})); -function adEpoch(c: any, screen = c.screen, state = c.binding.state, delta: any = {}) { - const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'permission-crop-replay-')); - const file = path.join(dir, 'state.json'); - try { - // Captures contain POSIX paths. Project their filesystem identity onto the - // replay host without changing the captured fixture or its menu rendering. - const nativeState = { ...state, cwd: path.resolve(state.cwd), expected: path.resolve(state.expected), - transcriptPath: path.resolve(state.transcriptPath) }; - const replayScreen = screen.replaceAll(path.posix.dirname(c.binding.expected), - path.dirname(path.resolve(c.binding.expected))); - fs.writeFileSync(file, JSON.stringify(nativeState)); - return currentFilePermissionEpoch(file, path.resolve(delta.expected ?? c.binding.expected), path.resolve(delta.cwd ?? c.p.cwd), - path.resolve(delta.config ?? c.p.config), delta.startedAt ?? c.p.startUnix * 1000, - delta.transcript ?? c.observation.transcript, replayScreen); - } finally { fs.rmSync(dir, {recursive:true, force:true}); } -} -for (const c of cases) { - test(`actual AD captured current permission ${c.p.pid} returns its exact pending epoch`, () => { - expect(classifyPlanCountFrame(c.screen)).toBe('permission'); - expect(adEpoch(c)).toEqual({pendingId:c.binding.state.pendingId, completedId:c.binding.state.completedId, - completedIds:c.binding.state.completedIds}); - const guard = createPlanCountPermissionGuard(); - expect(guard(c.screen, '', adEpoch(c))).toBe('grant'); - expect(guard(c.screen, '', adEpoch(c))).toBe('handled'); - }); - test(`AD isolating the rejected rendering guard ${c.p.pid} preserves native identity`, () => { - // These are explicitly normalized controls; the actual captured screen is unchanged above. - const normalized = c.p.pid === capturedAe.pid ? c.screen.replace(/^[╌─━]{3,}[ \t]*\n/, '') - : c.p.pid === 1332470 ? c.screen.replace('3. Nohift+tab)', '3. No') : c.screen.replace(/^ {4,5}\+/, ' 99 +'); - expect(normalized).not.toBe(c.screen); - expect(adEpoch(c, normalized)?.pendingId).toBe(c.binding.state.pendingId); - }); -} -for (const c of cases) { - test(`AD crop ${c.p.pid} rejects foreign, quoted, incomplete and policy-changing menus`, () => { - const directory = path.dirname(c.binding.expected); - const changes: [string, string][] = [ - ['foreign directory', c.screen.replace(directory, path.join(directory, 'foreign'))], - ['split directory', c.screen.replace(directory, directory + '\n/foreign')], - ['quoted', '> Example:\n' + c.screen], - ['AUQ', '☐ Review\n' + c.screen], - ['code fence', '```\n' + c.screen], - ['missing footer', c.screen.replace('Esc to cancel · Tab to amend', '')], - ['policy on selected Yes', c.screen.replace('❯ 1. Yes', '❯ 1. Yes, always allow')], - ['selected No', c.screen.replace('❯ 1. Yes', ' 1. Yes').replace(' 3. No', ' ❯ 3. No')], - ['arbitrary No suffix', c.screen.replace(/3\. No(?:hift\+tab\))?/, '3. No; run another command')], - ['another hint', c.screen.replace(/3\. No(?:hift\+tab\))?/, '3. No(shift+enter)')], - ['foreign option action', c.screen.replace(/always\s+allow\s+access\s+to/, 'delete files from')], - ['unrecognized cropped prose', c.screen.replace(/^.*\n/, 'arbitrary text\n')], - ]; - for (const [name, screen] of changes) { - expect(screen, name).not.toBe(c.screen); - expect(adEpoch(c, screen), name).not.toBeTruthy(); - expect(createPlanCountPermissionGuard()(screen, '', adEpoch(c, screen)), name).not.toBe('grant'); - } - }); - test(`AD crop ${c.p.pid} leaves unrelated-basename permission policy unchanged`, () => { - const screen = c.screen.replace(path.basename(c.binding.expected), 'OTHER.md'); - expect(adEpoch(c, screen)).toBeUndefined(); - // This is intentionally the existing caller policy for unrelated fixture permissions. - expect(createPlanCountPermissionGuard()(screen, '', adEpoch(c, screen))).toBe('grant'); - }); - test(`AD crop ${c.p.pid} cannot replace missing, stale, completed or foreign native identity`, () => { - const original = c.binding.state; - for (const [name, state, delta] of [ - ['foreign session', {...original, sessionId:'foreign'}, {}], - ['wrong native transcript', {...original, transcriptPath:path.join(c.p.config, 'projects', 'foreign', 'other.jsonl')}, {}], - ['completed request', {...original, completedId:original.pendingId}, {}], - ['no pending request', {...original, pendingId:null}, {}], - ['unseen pending request', {...original, pendingId:original.sessionId + ':other'}, {}], - ['stale timestamp', {...original, timestamp:new Date(c.p.startUnix * 1000 - 1).toISOString()}, {}], - ['future timestamp', {...original, timestamp:new Date(Date.now() + 60_000).toISOString()}, {}], - ['mixed sessions', original, {transcript:{status:'ready', calls:[], assistantMessages:[{sessionId:original.sessionId},{sessionId:'foreign'}]}}], - ['unready transcript', original, {transcript:{...c.observation.transcript,status:'unavailable'}}], - ] as const) { - expect(adEpoch(c, c.screen, state, delta), name).toBeNull(); - } - }); -} - -test('AD crop fixture selects the exact existing permission regression callers', () => { - expect(selectTests(['test/fixtures/plan-count-permission-ad.json'], E2E_TOUCHFILES).selected.sort()).toEqual( - selectTests(['test/fixtures/plan-count-permission-ac.json'], E2E_TOUCHFILES).selected.sort()); -}); - -test('AE crop admits one native divider only and preserves its exact existing caller selection', () => { - const c = cases.find(item => item.p.pid === capturedAe.pid)!; - const firstLine = c.screen.slice(0, c.screen.indexOf('\n') + 1); - for (const screen of [firstLine + c.screen, 'unrelated prose\n' + c.screen, - firstLine + 'Example:\n' + c.screen.slice(firstLine.length), - c.screen.replace(firstLine, firstLine.trimEnd() + ' extra action\n')]) { - expect(adEpoch(c, screen)).toBeNull(); - expect(createPlanCountPermissionGuard()(screen, '', adEpoch(c, screen))).not.toBe('grant'); - } - expect(selectTests(['test/fixtures/plan-count-permission-ae.json'], E2E_TOUCHFILES).selected.sort()).toEqual( - selectTests(['test/fixtures/plan-count-permission-ad.json'], E2E_TOUCHFILES).selected.sort()); -}); - -test('AH wrapped crop admits four or five spaces with the same owned native epoch', () => { - const c = cases.find(item => item.p.pid === capturedAh.pid)!; - expect(c.screen.startsWith(' + ')).toBe(true); - for (const screen of [c.screen, ` ${c.screen}`, c.screen.replace(/^ \+/, ' -')]) { - expect(adEpoch(c, screen)?.pendingId).toBe(c.binding.state.pendingId); - const guard = createPlanCountPermissionGuard(); - expect(guard(screen, '', adEpoch(c, screen))).toBe('grant'); - expect(guard(screen, '', adEpoch(c, screen))).toBe('handled'); - } - // Cleaning the unselected No paint residue does not establish missing identity. - const noOnly = c.screen.replace('3. Nohift+tab)', '3. No'); - expect(noOnly).not.toBe(c.screen); - expect(adEpoch(c, noOnly)?.pendingId).toBe(c.binding.state.pendingId); -}); - -test('AH continuation crop rejects prose, unsupported gutters and malformed numbered context', () => { - const c = cases.find(item => item.p.pid === capturedAh.pid)!; - for (const [name, screen] of [ - ['four-space prose', c.screen.replace(/^.*\n/, ' Apply this edit now\n')], - ['four-space quoted prose', c.screen.replace(/^.*\n/, ' > Example\n')], - ['three-space gutter', c.screen.slice(1)], - ['six-space gutter', ` ${c.screen}`], - ['no numbered rows', c.screen.replace(/^\s*\d+\s+(?=[+\- ])/gm, ' +')], - ['one numbered row', c.screen.replace(/^(\s*\d+\s+)(?=[+\- ])/gm, - (prefix, _group, offset) => offset === c.screen.indexOf(' 79 ') ? prefix : ' +')], - ]) { - expect(screen, name).not.toBe(c.screen); - expect(adEpoch(c, screen), name).toBeNull(); - expect(createPlanCountPermissionGuard()(screen, '', adEpoch(c, screen)), name).not.toBe('grant'); - } - expect(selectTests(['test/fixtures/plan-count-permission-ah.json'], E2E_TOUCHFILES).selected.sort()).toEqual( - selectTests(['test/fixtures/plan-count-permission-ae.json'], E2E_TOUCHFILES).selected.sort()); -}); diff --git a/test/plan-count-prerequisite-n.test.ts b/test/plan-count-prerequisite.test.ts similarity index 63% rename from test/plan-count-prerequisite-n.test.ts rename to test/plan-count-prerequisite.test.ts index 84e3a6f98..b063fc371 100644 --- a/test/plan-count-prerequisite-n.test.ts +++ b/test/plan-count-prerequisite.test.ts @@ -1,8 +1,99 @@ -import { describe, expect, test } from 'bun:test'; -import callFixture from './fixtures/ceo-prerequisite-n-call.json'; -import { capturePlanCountQuestion, nativePlanCallFingerprint, planCountPrerequisitePick } from './helpers/claude-pty-runner'; -import { E2E_TOUCHFILES } from './helpers/touchfiles-data'; +/** + * Native prerequisite and navigation picks (planCountPrerequisitePick) over captured calls. + */ +import { describe } from 'bun:test'; +import { expect } from 'bun:test'; +import { test } from 'bun:test'; +import { capturePlanCountQuestion } from './helpers/claude-pty-runner'; +import { nativePlanCallFingerprint } from './helpers/claude-pty-runner'; +import { planCountPrerequisitePick } from './helpers/claude-pty-runner'; +import { planCountQuestionInput } from './helpers/claude-pty-runner'; +import type { NativePlanQuestionCall } from './helpers/plan-count-transcript'; +import prerequisite_plan_count_navigation_r from './fixtures/dx-prerequisite-r-call.json'; +import callFixture_plan_count_prerequisite_n from './fixtures/ceo-prerequisite-n-call.json'; +import engPrerequisite77_plan_count_prerequisite_n from './fixtures/eng-prerequisite-77.json'; +describe('plan-count-navigation-r', () => { +const prerequisite = prerequisite_plan_count_navigation_r; +function pending(source: NativePlanQuestionCall): NativePlanQuestionCall { + const call = structuredClone(source); + call.answered = false; + delete call.answers; + delete call.unansweredQuestionIndices; + delete call.answeredAt; + return call; +} +function frame(call: NativePlanQuestionCall) { + const q = call.questions[0]!; + const visible = `☐ ${q.header}\n${q.question}\n${q.options.map((o, i) => `${i ? ' ' : '❯'} ${i + 1}. ${o.label}`).join('\n')}\nEnter to select · ↑/↓ to navigate · Esc to cancel`; + const active = capturePlanCountQuestion(visible, new Set(), 0, true, call)!; + return { visible, active, routing: nativePlanCallFingerprint(call, 0, true) }; +} + +describe('captured R planning navigation', () => { + test('DX declines its optional office-hours detour using the full native option meaning', () => { + const call = pending(prerequisite as NativePlanQuestionCall); + const { visible, active, routing } = frame(call); + expect(active.nativeCall).toBe(call); + expect(prerequisite.answers[prerequisite.questions[0]!.question]).toBe(prerequisite.questions[0]!.options[0]!.label); + const pick = planCountPrerequisitePick(routing, active) ?? 1; + expect(pick).toBe(2); + expect(planCountQuestionInput(visible, active, pick)).toBe('2'); + }); + test('short prerequisite labels require the current native question and affirmative review action', () => { + for (const change of [ + (c: NativePlanQuestionCall) => { c.questions[0]!.options[1]!.description = ''; }, + (c: NativePlanQuestionCall) => { c.questions[0]!.options[1]!.description = 'Do not proceed with standard DX POLISH review.'; }, + (c: NativePlanQuestionCall) => { c.questions[0]!.options[1]!.description = 'Accept the finding and proceed with standard DX POLISH review.'; }, + (c: NativePlanQuestionCall) => { c.questions[0]!.options[1]!.description = 'Accept this security finding. Proceed with standard DX POLISH review.'; }, + (c: NativePlanQuestionCall) => { c.questions[0]!.options[1]!.description = 'Proceed with standard DX POLISH review after running /office-hours first.'; }, + (c: NativePlanQuestionCall) => { c.questions[0]!.options[1]!.description = 'Proceed with standard DX POLISH review? No, run /office-hours first.'; }, + (c: NativePlanQuestionCall) => { c.questions[0]!.options[1]!.description = 'Proceed with standard DX POLISH review. Accept the security risk.'; }, + (c: NativePlanQuestionCall) => { c.questions[0]!.options[1]!.description = 'Plan scope is already precise. Proceed with standard DX POLISH review if the tests pass.'; }, + (c: NativePlanQuestionCall) => { c.questions[0]!.options.push({ label: 'Accept this security finding' }); }, + (c: NativePlanQuestionCall) => { c.questions[0]!.multiSelect = true; }, + (c: NativePlanQuestionCall) => { c.questions.push(structuredClone(c.questions[0]!)); }, + (c: NativePlanQuestionCall) => { c.answered = true; }, + (c: NativePlanQuestionCall) => { c.failed = true; }, + ]) { + const call = pending(prerequisite as NativePlanQuestionCall); + change(call); + const fp = nativePlanCallFingerprint(call, 0, true); + expect(planCountPrerequisitePick(fp)).toBeNull(); + } + const call = pending(prerequisite as NativePlanQuestionCall); + const { active, routing, visible } = frame(call); + expect(planCountPrerequisitePick(routing, { ...active, preReview: false })).toBeNull(); + expect(planCountPrerequisitePick(routing, { ...active, signature: 'stale:question' })).toBeNull(); + expect(planCountPrerequisitePick(routing, { ...active, nativeQuestionIndex: 1 })).toBeNull(); + expect(planCountPrerequisitePick(routing, { ...active, options: active.options.slice().reverse() })).toBeNull(); + const visibleOnly = capturePlanCountQuestion(visible, new Set(), 0, true)!; + expect(visibleOnly.nativeCall).toBeUndefined(); + expect(planCountPrerequisitePick(routing, visibleOnly)).toBeNull(); + call.questions[0]!.options.reverse(); + const reordered = frame(call); + expect(planCountPrerequisitePick(reordered.routing, reordered.active)).toBe(1); + }); + + test('a bare short label uses its bound unconditional review meaning', () => { + const call = pending(prerequisite as NativePlanQuestionCall);call.questions[0]!.options[1]!.label='Skip'; + const {active,routing}=frame(call);expect(planCountPrerequisitePick(routing,active)).toBe(2); + }); + + test('short labels allow only a benign plan-scope premise plus the unconditional review action', () => { + for (const description of ['Proceed with standard review.', 'Proceed with standard DX review', 'The plan is clear. Proceed with standard DX POLISH review.', 'Plan scope is already precise. Proceed with standard DX POLISH review.']) { + const call = pending(prerequisite as NativePlanQuestionCall); + call.questions[0]!.options[1]!.description = description; + const { active, routing } = frame(call); + expect(planCountPrerequisitePick(routing, active)).toBe(2); + } + }); +}); +}); + +describe('plan-count-prerequisite-n', () => { +const callFixture = callFixture_plan_count_prerequisite_n; +const engPrerequisite77 = engPrerequisite77_plan_count_prerequisite_n; const call = () => structuredClone(callFixture); describe('native prerequisite review-now offer', () => { @@ -45,27 +136,8 @@ describe('native prerequisite review-now offer', () => { expect(planCountPrerequisitePick(nativePlanCallFingerprint(native, 0, true))).toBeNull(); } }); - - test('the captured prerequisite regression selects its exact counting and mode consumers', () => { - // Floor checks also seed a plan, but never pick a prerequisite answer. - const expected = [ - 'plan-ceo-mode-routing', 'plan-ceo-split-overflow', 'plan-design-with-ui-scope', 'plan-eng-multi-finding-batching', - ].sort(); - for (const dependency of ['test/plan-count-prerequisite-n.test.ts', 'test/fixtures/ceo-prerequisite-n-call.json', 'test/fixtures/eng-prerequisite-77.json']) { - const consumers = Object.entries(E2E_TOUCHFILES).filter(([, paths]) => paths.includes(dependency)); - expect(consumers.map(([name]) => name).sort()).toEqual(expected); - for (const [, paths] of consumers) { - expect(paths).toContain('test/helpers/plan-count-fixture.ts'); - expect(paths).toContain('test/plan-count-fixture.test.ts'); - } - } - }); }); - -import engPrerequisite77 from './fixtures/eng-prerequisite-77.json'; -import type { NativePlanQuestionCall } from './helpers/plan-count-transcript'; - function engPrerequisitePending(): NativePlanQuestionCall { const native = structuredClone(engPrerequisite77.completedCall) as NativePlanQuestionCall; native.answered = false; @@ -177,3 +249,4 @@ describe('optional Office Hours decision briefs', () => { expect(planCountPrerequisitePick(other.routing, other.active)).toBeNull(); }); }); +}); diff --git a/test/plan-count-quoted-frame-ak.test.ts b/test/plan-count-quoted-frame-ak.test.ts deleted file mode 100644 index 632d18e58..000000000 --- a/test/plan-count-quoted-frame-ak.test.ts +++ /dev/null @@ -1,106 +0,0 @@ -import {expect,test} from 'bun:test'; -import fs from 'node:fs'; -import os from 'node:os'; -import path from 'node:path'; -import {pathToFileURL} from 'node:url'; -import {classifyPlanCountFrame,createPlanCountPermissionGuard} from './helpers/claude-pty-runner'; -import exact from './fixtures/plan-count-quoted-frame-ak.json'; -import native from './fixtures/plan-count-permission-ac.json'; -import owned from './fixtures/plan-count-owned-permission-v.json'; -import {E2E_TOUCHFILES,selectTests} from './helpers/touchfiles'; -const quote=(s:string)=>s.split('\n').map(row=>'> '+row).join('\n'); - -// A PTY transports bytes, not command-sized stdin events. Share the framing -// code with the fake CLI so fragmented grants exercise the same receiver. -function commandBuffer(){ - let pending=''; - return (chunk:string)=>{ - pending+=chunk;const commands:string[]=[];let end:number; - while((end=pending.indexOf('\r'))!==-1){commands.push(pending.slice(0,end+1));pending=pending.slice(end+1);} - return commands; - }; -} -test('fake CLI preserves command bytes across fragmented and coalesced PTY input',()=>{ - const expected=['/plan-ceo-review\r','1\r']; - for(const chunks of [expected,['/plan-ceo-review\r','1','\r'],[...expected.join('')],[expected.join('')]]){ - const receive=commandBuffer();expect(chunks.flatMap(receive)).toEqual(expected); - } - const receive=commandBuffer(); - expect(receive('1')).toEqual([]);expect(receive('\r2\rtrailing')).toEqual(['1\r','2\r']); - expect(receive('\r')).toEqual(['trailing\r']); // no unexpected bytes are discarded -}); - -test('the exact wholly quoted AK pane is handled so the dispatcher sends no fallback',()=>{ - const screen=quote(exact.screen); - expect(classifyPlanCountFrame(screen)).toBe('permission'); - expect(createPlanCountPermissionGuard()(screen,'',undefined)).toBe('handled'); -}); -test('quoted whole native frames remain inert across redraw, indentation, CRLF and terminal color',()=>{ - for(const source of [exact.screen,native.rows[0]!.screen,owned.screen])for(const transform of [ - (s:string)=>quote(s),(s:string)=>'\n'+quote(s)+'\n',(s:string)=>quote(s).replace(/^>/gm,' >'), - (s:string)=>quote(s).replaceAll('\n','\r\n'),(s:string)=>'\x1b[31m'+quote(s)+'\x1b[0m', - ]){const s=transform(source),g=createPlanCountPermissionGuard(); - const expected=classifyPlanCountFrame(s)==='permission'?'handled':null; - expect(g(s,'')).toBe(expected);expect(g(s,'⎿ Wrote 4 lines\n')).toBe(expected);} -}); -test('quoted history cannot consume or authorize the current native grant',()=>{ - const s=native.rows[0]!.screen,g=createPlanCountPermissionGuard(),q=quote(s); - expect(g(q,'')).toBe('handled');expect(g(s,q)).toBe('grant');expect(g(q,s)).toBe('handled');expect(g(s,q)).toBe('handled'); - expect(createPlanCountPermissionGuard()(s,'',null)).toBe('handled'); - expect(createPlanCountPermissionGuard()(s,'',{pendingId:'main:first',completedId:null})).toBe('grant'); - expect(createPlanCountPermissionGuard()(q,'',{pendingId:'main:first',completedId:null})).toBe('handled'); -}); -test('unquoted current native frames retain their policy even with quote characters in diff text or history',()=>{ - for(const s of native.rows.map(row=>row.screen))expect(createPlanCountPermissionGuard()(s,quote(s))).toBe('grant'); - expect(createPlanCountPermissionGuard()('> An old note\n'+native.rows[0]!.screen,'')).toBe('grant'); - expect(createPlanCountPermissionGuard()('No current pane',quote(exact.screen))).toBeNull(); - expect(createPlanCountPermissionGuard()('> Ordinary quoted prose, without a permission menu')).toBeNull(); -}); -test('the regression selects exactly the existing permission consumers',()=>{ - for(const dependency of ['test/plan-count-quoted-frame-ak.test.ts','test/fixtures/plan-count-quoted-frame-ak.json']) - expect(selectTests([dependency],E2E_TOUCHFILES).selected.sort()).toEqual(selectTests(['test/plan-count-permission-ac.test.ts'],E2E_TOUCHFILES).selected.sort()); -}); - -test.skipIf(process.platform==='win32')('real dispatcher ignores quoted pane then grants the fresh owned native request once',async()=>{ - const dir=fs.mkdtempSync(path.join(os.tmpdir(),'count-quoted-frame-')),fake=path.join(dir,'fake-claude'),worker=path.join(dir,'worker.ts'),events=path.join(dir,'events.jsonl'),output=path.join(dir,'result.json'),report=path.join(dir,'report.md'); - fs.writeFileSync(report,'original'); - fs.writeFileSync(fake,`#!${process.execPath}\nconst receive=(${commandBuffer.toString()})();\n`+String.raw` -import fs from 'node:fs';import path from 'node:path'; -const item=JSON.parse(process.env.QUOTED_FRAME_CASE),sid='quoted-frame-main',log=e=>fs.appendFileSync(item.events,JSON.stringify(e)+'\n'); -const transcript=path.join(process.env.CLAUDE_CONFIG_DIR,'projects','owned',sid+'.jsonl');fs.mkdirSync(path.dirname(transcript),{recursive:true}); -const native=(role,content,extra={})=>fs.appendFileSync(transcript,JSON.stringify({cwd:process.cwd(),sessionId:sid,isSidechain:false,timestamp:new Date().toISOString(),message:{role,content},...extra})+'\n'); -native('assistant',[{type:'text',text:'Reviewing the owned fixture.'}]);log({type:'start',pid:process.pid,cwd:process.cwd()}); -const settings=JSON.parse(process.argv[process.argv.indexOf('--settings')+1]); -const hook=async name=>{for(const entry of settings.hooks[name]??[]){if(entry.matcher!=='^(Write|Edit)$')continue; - const event={hook_event_name:name,tool_name:'Edit',session_id:sid,tool_use_id:'current',cwd:process.cwd(),transcript_path:transcript,tool_input:{file_path:item.report}}; - const p=Bun.spawn(['bash','-c',entry.hooks[0].command],{stdin:new Blob([JSON.stringify(event)]),stdout:'pipe',stderr:'pipe'}); - const [code,out,err]=await Promise.all([p.exited,new Response(p.stdout).text(),new Response(p.stderr).text()]);if(code||out||err)throw Error('Hook failed');}}; -const pane=item.screen.replaceAll('PLAN.md',item.report),paint=s=>process.stdout.write('\x1b[2J\x1b[H'+s.replaceAll('\n','\r\n')); -let stage='startup';process.stdin.setRawMode?.(true);const dispatch=async input=>{ - log({type:'input',stage,input}); - if(stage==='startup'){stage='quoted';paint(item.quotedScreen);setTimeout(async()=>{await hook('PreToolUse');stage='current';paint(pane);},4200);return;} - if(stage!=='current'){log({type:'unexpected'});return;} - if(input!=='1\r')throw Error('One-time grant changed');stage='done';await hook('PostToolUse'); - const q={header:'Finding',question:'Apply the reviewed fix?',options:[{label:'Fix'},{label:'Keep'}]}; - native('assistant',[{type:'tool_use',name:'AskUserQuestion',id:'finding',input:{questions:[q]}}]);native('user',[{type:'tool_result',tool_use_id:'finding',content:'Answered'}],{toolUseResult:{answers:{[q.question]:'Fix'}}});paint('Done.\n'); -};process.stdin.on('data',async data=>{const chunk=data.toString();log({type:'chunk',stage,input:chunk});for(const input of receive(chunk))await dispatch(input);});process.on('SIGINT',()=>process.exit(0));process.stdin.resume(); -process.stdout.write('PTY_READY:'+item.events+'\x1b[2J\x1b[H'); -`);fs.chmodSync(fake,0o755); - // Keep every physical terminal row inside the quote; adding a prefix to an - // already120-column capture would otherwise wrap an unquoted continuation. - const quotedScreen=exact.screen.split('\n').flatMap(row=>row.trimEnd().match(/.{1,116}/gu)??['']).map(row=>'> '+row).join('\n'); - const args={skillName:'plan-ceo-review',slashCommand:'/plan-ceo-review',followUpPrompt:'Review the disposable fixture.',expectedPlanPath:report,reviewCountCeiling:1,timeoutMs:25000,startupReadyMarker:'PTY_READY:'+events,env:{QUOTED_FRAME_CASE:JSON.stringify({events,report,screen:owned.screen,quotedScreen})}}; - fs.writeFileSync(worker,`import {runPlanSkillCounting} from ${JSON.stringify(pathToFileURL(path.join(import.meta.dir,'helpers/claude-pty-runner.ts')).href)};const result=await runPlanSkillCounting({...${JSON.stringify(args)},isLastStep0AUQ:()=>false,isReviewAUQ:()=>true});await Bun.write(${JSON.stringify(output)},JSON.stringify(result));`); - const child=Bun.spawn([process.execPath,worker],{env:{...process.env,BROWSE_TERMINAL_BINARY:fake,EVALS_HERMETIC:'1'},stdout:'pipe',stderr:'pipe'}),timer=setTimeout(()=>child.kill('SIGKILL'),30000); - try{const [code,out,err]=await Promise.all([child.exited,new Response(child.stdout).text(),new Response(child.stderr).text()]);expect(code,out+err).toBe(0); - const result=JSON.parse(fs.readFileSync(output,'utf8')),rows=fs.readFileSync(events,'utf8').trim().split('\n').map(s=>JSON.parse(s)); - expect(result.outcome,JSON.stringify({result,rows})).toBe('ceiling_reached');expect(result.reviewCount).toBe(1); - const chunks=rows.filter(r=>r.type==='chunk'); - expect(chunks.every(r=>['startup','current'].includes(r.stage))).toBe(true); - expect(chunks.map(r=>r.input).join('')).toBe('/plan-ceo-review\r1\r'); - expect(rows.filter(r=>r.type==='input').map(r=>[r.stage,r.input])).toEqual([['startup','/plan-ceo-review\r'],['current','1\r']]);expect(rows.some(r=>r.type==='unexpected')).toBe(false); - expect(()=>process.kill(rows[0].pid,0)).toThrow();expect(fs.existsSync(rows[0].cwd)).toBe(false); - }finally{clearTimeout(timer);child.kill('SIGKILL');await child.exited; - if(fs.existsSync(events)){const first=JSON.parse(fs.readFileSync(events,'utf8').split('\n')[0]!);try{if(fs.readFileSync('/proc/'+first.pid+'/cmdline','utf8').split('\0').includes(fake))process.kill(first.pid,'SIGKILL');}catch{}} - fs.rmSync(dir,{recursive:true,force:true});} -},32000); diff --git a/test/plan-scope-recovery-av.test.ts b/test/plan-scope-recovery-av.test.ts deleted file mode 100644 index 9c32bb325..000000000 --- a/test/plan-scope-recovery-av.test.ts +++ /dev/null @@ -1,97 +0,0 @@ -import {expect, test} from 'bun:test'; -import fs from 'node:fs'; -import path from 'node:path'; -import {ALL_HOST_CONFIGS} from '../hosts'; -import {generatePreamble} from '../scripts/resolvers/preamble'; -import {HOST_PATHS, type TemplateContext} from '../scripts/resolvers/types'; -import {nativeSeededPlanSelection} from './helpers/plan-scope-selection'; -import {E2E_TOUCHFILES, LLM_JUDGE_TOUCHFILES, selectTests} from './helpers/touchfiles'; -import observedFailures from './fixtures/plan-scope-recovery-av.json'; - -const skills = ['plan-eng-review', 'plan-design-review'] as const; -const read = (skill: string) => fs.readFileSync(path.join(import.meta.dir, '..', skill, 'SKILL.md.tmpl'), 'utf8'); -const recovery = (text: string) => text.split('\n').find(line => line.startsWith('> Before ') && line.includes('publicly identified'))!; - -test('the review handoff repairs a missing public declaration without claiming timely compliance', () => { - for (const skill of skills) { - const text = read(skill), check = recovery(text); - expect(check).toBeDefined(); - expect(check).toContain('require resolved scope'); - expect(check).toContain('For plan-mode auto-selection, verify you publicly identified the selected plan for this invocation before review work'); - expect(check).toContain('If missing, send "Scope gate: plan mode — auto-selected B (reviewing )." now'); - expect(check).toContain('do not claim an earlier announcement'); - const start = skill === 'plan-eng-review' ? '### Step 0: Scope Challenge' : '## PRE-REVIEW SYSTEM AUDIT'; - expect(text.indexOf(check)).toBeGreaterThan(text.indexOf('{{PREAMBLE}}')); - expect(text.indexOf(check)).toBeGreaterThan(text.indexOf(start)); - const reviewStart = text.indexOf(skill === 'plan-eng-review' ? '{{SECTION:review-sections}}' : 'Before reviewing the plan, gather context'); - expect(reviewStart).toBeGreaterThanOrEqual(0); - expect(text.indexOf(check)).toBeLessThan(reviewStart); - if (skill === 'plan-eng-review') { - const section = fs.readFileSync(path.join(import.meta.dir, '..', skill, 'sections/review-sections.md.tmpl'), 'utf8'); - expect(section).toContain('### A. Assess the target'); - expect(section).toContain('Complete these checks before the complexity decision in B'); - expect(section.indexOf('### A. Assess the target')).toBeLessThan(section.indexOf('### B. Resolve complexity selectors')); - expect(section.indexOf('### B. Resolve complexity selectors')).toBeLessThan(section.indexOf('### C. Resolve findings')); - expect(section).toContain('Run C whether B was completed or skipped'); - expect(text.slice(text.indexOf(check), reviewStart)).toContain('Scope Challenge is mandatory before Section 1'); - } - } -}); - -test('unseeded, explicit-target and early announcement rules remain authoritative on every host', () => { - for (const skill of skills) { - const template = read(skill); - const gate = template.slice(template.indexOf('## Scope gate'), template.indexOf('{{PREAMBLE}}')); - const entry = skill === 'plan-eng-review' - ? 'Before tools or preamble, resolve from provided messages, listed tools and explicit host metadata only' - : 'After this skill loads, resolve this gate before any tool'; - const announce = skill === 'plan-eng-review' - ? 'Announce an auto-selected plan in one line so the user can interrupt' - : 'Announce plan-mode auto-selection before review tools'; - expect(gate).toContain(entry); - expect(gate).toContain(announce); - expect(gate).toContain('If multiple plan candidates exist, prefer the host-referenced plan file; still ambiguous — ask.'); - expect(gate).toContain('If the user explicitly named a DIFFERENT target'); - expect(gate).toContain('If plan mode is indicated but no plan exists yet, ask as normal'); - expect(gate).toContain('When no exception above applied:'); - expect(gate).toContain(skill === 'plan-eng-review' - ? 'First tool call = AskUserQuestion (tool_use). Send this exact menu and wait' - : 'First tool call = AskUserQuestion (tool_use). Confirm what to review.'); - expect(gate).toContain('STOP and wait for the answer'); - for (const host of ALL_HOST_CONFIGS) { - const ctx: TemplateContext = {skillName: skill, tmplPath: `${skill}/SKILL.md.tmpl`, host: host.name, - paths: HOST_PATHS[host.name]!, preambleTier: 3, interactive: true}; - const expanded = template.replace('{{PREAMBLE}}', generatePreamble(ctx)); - expect(expanded.indexOf(announce)).toBeLessThan(expanded.indexOf('```bash')); - expect(expanded.indexOf('STOP and wait for the answer')).toBeLessThan(expanded.indexOf('```bash')); - expect(expanded.indexOf(recovery(template))).toBeGreaterThan(expanded.indexOf('```bash')); - } - } -}); - -test('recorded failures stay intact while fresh unique-draft introductions now bind', () => { - expect(observedFailures).toHaveLength(4); - for (const [index, row] of observedFailures.entries()) { - expect(row.observed.scopeGateAutoSelectObserved).toBe(false); - expect(nativeSeededPlanSelection(row.transcript as any, row.tools as any, row.opts)).toBe(index !== 0); - const title = /^#\s+(?:Plan:\s*)?(.+)$/m.exec(row.opts.seed)![1]!; - const loaded = row.tools.find(tool => tool.kind === 'result')!; - const timestamp = new Date(Date.parse(loaded.timestamp) + 1).toISOString(); - const message = {sessionId: row.opts.sessionId, timestamp, text: `I'll review "${title}" plan.`}; - const amended = {...row.transcript, assistantMessages: [message]}; - // Explicit synthetic control only: no changed message is paid evidence. - expect(nativeSeededPlanSelection(amended as any, row.tools as any, row.opts)).toBe(true); - for (const text of [`> ${message.text}`, `Example:\n${message.text}`, `I'll review "Another plan" plan.`]) { - expect(nativeSeededPlanSelection({...amended, assistantMessages: [{...message, text}]} as any, row.tools as any, row.opts)).toBe(false); - } - } -}); - -test('regression sources select exactly the union of the two template owners', () => { - for (const map of [E2E_TOUCHFILES, LLM_JUDGE_TOUCHFILES]) { - const expected = selectTests(skills.map(skill => `${skill}/SKILL.md.tmpl`), map, []).selected; - for (const file of ['test/plan-scope-recovery-av.test.ts', 'test/fixtures/plan-scope-recovery-av.json']) { - expect(selectTests([file], map, []).selected).toEqual(expected); - } - } -}); diff --git a/test/plan-scope-selection.test.ts b/test/plan-scope-selection.test.ts index 8319ef4d9..b99a927e0 100644 --- a/test/plan-scope-selection.test.ts +++ b/test/plan-scope-selection.test.ts @@ -11,6 +11,21 @@ import { nativeSeededPlanSelection } from './helpers/plan-scope-selection'; import { isScopeGateQuestionVisible } from './helpers/claude-pty-runner'; import { readPlanCountTranscript, type NativePublicToolEvent, type PlanCountTranscript } from './helpers/plan-count-transcript'; import { selectTests, E2E_TOUCHFILES } from './helpers/touchfiles'; +import captured_design_scope_announcement_ao from './fixtures/design-scope-announcement-ao.json'; +import fixture_design_scope_declaration_ak from './fixtures/design-scope-declaration-ak.json'; +import fs_design_scope_entry_aq from 'node:fs'; +import path_design_scope_entry_aq from 'node:path'; +import { ALL_HOST_CONFIGS } from '../hosts'; +import { HOST_PATHS } from '../scripts/resolvers/types'; +import type { TemplateContext } from '../scripts/resolvers/types'; +import { generatePreamble } from '../scripts/resolvers/preamble'; +import { generateBaseBranchDetect } from '../scripts/resolvers/utility'; +import failedScopes_design_scope_entry_aq from './fixtures/design-scope-checkpoint-at.json'; +import { isScopeGateAutoSelectVisible } from './helpers/claude-pty-runner'; +import capture_design_scope_selection_aj from './fixtures/design-scope-selection-aj.json'; +import fixture_eng_option_b_scope_al from './fixtures/eng-option-b-scope-al.json'; +import observedFailures_plan_scope_recovery_av from './fixtures/plan-scope-recovery-av.json'; +import { describe } from 'bun:test'; const START = Date.parse('2026-09-10T00:25:00Z'); const opts = { seed: '# Plan: Marketing landing page\n\n## Layout\nA draft.', skillName: 'plan-design-review', sessionId: 'owned', commandStartedAt: START }; @@ -426,3 +441,520 @@ test('spoken skill names retain scope ownership, currentness and target boundari expect(verdict(spokenEngFixture(text),{...engScope,seed:opts.seed+'\n# Another plan'})).toBe(false); } }); + +describe('design-scope-announcement-ao', () => { +const captured = captured_design_scope_announcement_ao; +const input = () => structuredClone(captured.projection); +type Input = ReturnType; +const announcement = (p: Input) => p.transcript.assistantMessages.find(m => m.sessionId === p.opts.sessionId && m.text.startsWith("I'll auto-select"))!; +const verdict = (p: Input) => nativeSeededPlanSelection(p.transcript as PlanCountTranscript, p.tools as NativePublicToolEvent[], p.opts); + +test('the exact owned post-load option B announcement selects the seeded title', () => { + expect(captured.rawScopeGateAutoSelectObserved).toBe(false); + expect(verdict(input())).toBe(true); +}); + +test('equivalent explicit selection words and balanced title quotes retain identity', () => { + for (const prefix of ["I'll auto-select", 'I will auto-select', "I'll auto select"]) { + for (const title of ['Marketing landing page', '"Marketing landing page"', '“Marketing landing page”', '`Marketing landing page`']) { + const p = input(), m = announcement(p); + m.text = m.text.replace("I'll auto-select", prefix).replace('Marketing landing page', title); + expect(verdict(p)).toBe(true); + } + } + const p = input(), m = announcement(p); + p.opts.seed = p.opts.seed.replace('Marketing landing page', 'Account settings'); + m.text = m.text.replace('Marketing landing page', 'Account settings'); + expect(verdict(p)).toBe(true); +}); + +const rejected: Array<[string, (p: Input) => void]> = [ + ['wrong option', p => { announcement(p).text = announcement(p).text.replace('option B', 'option A'); }], + ['wrong target', p => { announcement(p).text = announcement(p).text.replace('Marketing landing page', 'Account settings'); }], + ['target prefix only', p => { announcement(p).text = announcement(p).text.replace('page draft', 'page experiment draft'); }], + ['conditional selection', p => { announcement(p).text = 'If approved: ' + announcement(p).text; }], + ['source selection', p => { announcement(p).text = 'Source excerpt:\n' + announcement(p).text; }], + ['quoted selection', p => { announcement(p).text = '> ' + announcement(p).text; }], + ['wholly quoted selection', p => { announcement(p).text = '"' + announcement(p).text + '"'; }], + ['unbalanced target quotes', p => { announcement(p).text = announcement(p).text.replace('Marketing landing page', '"Marketing landing page'); }], + ['question instead of assertion', p => { announcement(p).text = announcement(p).text.replace(/\.$/, '?'); }], + ['conditional tail', p => { announcement(p).text = announcement(p).text.replace(', running', ' if approved, running'); }], + ['cancelled selection', p => { announcement(p).text += '\nCorrection: this selection is withdrawn.'; }], + ['quoted status cancellation', p => { announcement(p).text += '\nThis selection is "withdrawn".'; }], + ['replaced target', p => { announcement(p).text += '\nThe selected target is now the branch diff.'; }], + ['pre-invocation announcement', p => { announcement(p).timestamp = new Date(p.opts.commandStartedAt - 1).toISOString(); }], + ['foreign announcement', p => { announcement(p).sessionId = 'foreign'; }], + ['foreign load result', p => { p.tools[1]!.sessionId = 'foreign'; }], + ['failed skill load', p => { p.tools[1]!.isError = true; }], + ['wrong skill', p => { p.tools[0]!.input!.skill = 'plan-eng-review'; }], + ['late command start', p => { p.opts.commandStartedAt = Date.parse(p.tools[1]!.timestamp) + 1; }], + ['multiple seed titles', p => { p.opts.seed += '\n# Another plan\n'; }], +]; +test.each(rejected)('%s supplies no scope selection', (_, change) => { + const p = input(); p.transcript.assistantMessages = [announcement(p)]; change(p); expect(verdict(p)).toBe(false); +}); + +test('quoted historical or foreign withdrawals do not replace the current selection', () => { + for (const correction of ['> This selection is withdrawn.', 'Historical note: "This selection is withdrawn."']) { + const p = input(); announcement(p).text += '\n' + correction; expect(verdict(p)).toBe(true); + } + const p = input(), m = announcement(p); + p.transcript.assistantMessages.push({ ...m, sessionId: 'foreign', text: 'This selection is withdrawn.' }); + expect(verdict(p)).toBe(true); +}); + +test('a later current withdrawal invalidates selection until a later reselection', () => { + const p = input(), m = announcement(p); + p.transcript.assistantMessages.push({ ...m, timestamp: new Date(Date.parse(m.timestamp) + 1000).toISOString(), text: 'This selection is withdrawn.' }); + expect(verdict(p)).toBe(false); + p.transcript.assistantMessages.push({ ...m, timestamp: new Date(Date.parse(m.timestamp) + 2000).toISOString() }); + expect(verdict(p)).toBe(true); +}); +}); + +describe('design-scope-declaration-ak', () => { +const fixture = fixture_design_scope_declaration_ak; +const input = (attempt = 0) => structuredClone(fixture.attempts[attempt]!.projection); +const verdict = (p = input()) => nativeSeededPlanSelection(p.transcript as PlanCountTranscript, p.tools as NativePublicToolEvent[], p.opts); +const declaration = (p: ReturnType) => p.transcript.assistantMessages.find(m => /^(?:I'll proceed with reviewing|Scope gate confirms plan mode)/.test(m.text))!; + +test('both exact owned post-load announcements select the named pasted draft', () => { + for (let attempt = 0; attempt < 2; attempt++) { + const p = input(attempt); + expect(fixture.attempts[attempt]!.rawScopeGateAutoSelectObserved).toBe(false); + expect(verdict(p)).toBe(true); + } +}); + +test('the prior AJ fresh unique-draft introduction now binds without changing its recorded outcome', () => { + const p = fixture.priorGenuineFailure.projection; + expect(nativeSeededPlanSelection(p.transcript as PlanCountTranscript, p.tools as NativePublicToolEvent[], p.opts)).toBe(true); +}); + +test('target identity and ordinary equivalent current review wording remain bound', () => { + for (let attempt = 0; attempt < 2; attempt++) { + const p = input(attempt); p.opts.seed = p.opts.seed.replace('Marketing landing page', 'Account settings'); + p.transcript.assistantMessages.forEach(m => { m.text = m.text.replaceAll('Marketing landing page', 'Account settings'); }); + for (const t of p.tools) if (t.input?.args) t.input.args = t.input.args.replaceAll('Marketing landing page', 'Account settings'); + expect(verdict(p)).toBe(true); + } + const p = input(); declaration(p).text = declaration(p).text.replace("I'll proceed", 'I will proceed'); expect(verdict(p)).toBe(true); +}); + +test('source, historical, quoted, hypothetical and conditional introductions do not select', () => { + for (let attempt = 0; attempt < 2; attempt++) for (const prefix of [ + '> ', ' ', 'Source excerpt:\n', 'Historical example only.\n', 'The following is hypothetical. ', 'If approved, ', '```\n', '"', + ]) { + const p = input(attempt), m = declaration(p); p.transcript.assistantMessages = [m]; m.text = prefix + m.text; + expect(verdict(p)).toBe(false); + } +}); + +test('a different target or conditional scope announcement cannot borrow the draft name', () => { + for (let attempt = 0; attempt < 2; attempt++) for (const change of [ + (s: string) => s.replaceAll('Marketing landing page', 'Checkout redesign'), + (s: string) => s.replace(/draft(?: plan)?/, 'draft plan if approved'), + ]) { const p = input(attempt), m = declaration(p); p.transcript.assistantMessages = [m]; m.text = change(m.text); expect(verdict(p)).toBe(false); } + for (const prefix of ['Scope gate might confirm plan mode, so', 'Scope gate confirms branch mode, so']) { + const p = input(1); p.transcript.assistantMessages = [declaration(p)]; declaration(p).text = declaration(p).text.replace('Scope gate confirms plan mode, so', prefix); expect(verdict(p)).toBe(false); + } +}); + +test('the same successful Skill load and post-command current session remain necessary', () => { + for (let attempt = 0; attempt < 2; attempt++) for (const change of [ + (p: ReturnType) => { p.opts.sessionId = 'foreign'; }, + (p: ReturnType) => { p.tools[1]!.isError = true; }, + (p: ReturnType) => { p.tools[1]!.toolUseId = 'foreign'; }, + (p: ReturnType) => { p.tools[0]!.input!.skill = 'plan-eng-review'; }, + (p: ReturnType) => { p.opts.commandStartedAt = Date.parse(p.tools[1]!.timestamp) + 1; }, + (p: ReturnType) => { declaration(p).timestamp = new Date(p.opts.commandStartedAt - 1).toISOString(); p.transcript.assistantMessages = [declaration(p)]; }, + ]) { const p = input(attempt); change(p); expect(verdict(p)).toBe(false); } +}); + +test('same-message and later current withdrawals or replacement targets defeat selection', () => { + for (let attempt = 0; attempt < 2; attempt++) for (const correction of [ + 'Correction: this selection is withdrawn.', + 'The selected target is now the branch diff.', + 'This declaration has been retracted.', + ]) for (const later of [false, true]) { + const p = input(attempt), m = declaration(p); p.transcript.assistantMessages = [m]; + if (later) p.transcript.assistantMessages.push({ ...m, timestamp: new Date(Date.parse(m.timestamp) + 1000).toISOString(), text: correction }); + else m.text += '\n' + correction; + expect(verdict(p)).toBe(false); + } +}); + +test('literal or foreign corrections preserve the actual declaration and a later reselection is current', () => { + for (let attempt = 0; attempt < 2; attempt++) { + const p = input(attempt), m = declaration(p); p.transcript.assistantMessages = [m]; + p.transcript.assistantMessages.push({ ...m, sessionId: 'foreign', text: 'The selected target is now the branch diff.' }); + p.transcript.assistantMessages.push({ ...m, text: '> This selection is withdrawn.' }); expect(verdict(p)).toBe(true); + p.transcript.assistantMessages.push({ ...m, timestamp: new Date(Date.parse(m.timestamp) + 1000).toISOString(), text: 'This selection is withdrawn.' }); expect(verdict(p)).toBe(false); + p.transcript.assistantMessages.push({ ...m, timestamp: new Date(Date.parse(m.timestamp) + 2000).toISOString() }); expect(verdict(p)).toBe(true); + } +}); +}); + +describe('design-scope-entry-aq', () => { +const fs = fs_design_scope_entry_aq; +const path = path_design_scope_entry_aq; +const failedScopes = failedScopes_design_scope_entry_aq; +const template = fs.readFileSync(path.join(import.meta.dir, '../plan-design-review/SKILL.md.tmpl'), 'utf8'); +const scope = template.slice(template.indexOf('## Scope gate'), template.indexOf('## Design Philosophy')); +const announcement = 'Scope gate: plan mode — auto-selected B (reviewing ).'; + +test('Design resolves scope before either executable bootstrap placeholder', () => { + const gate = template.indexOf('## Scope gate'); + expect(gate).toBeGreaterThan(0); + for (const token of ['{{PREAMBLE}}', '{{BASE_BRANCH_DETECT}}']) { + expect(template.split(token)).toHaveLength(2); + expect(template.indexOf(announcement)).toBeLessThan(template.indexOf(token)); + expect(template.indexOf('Reply with A, B, or C. STOP and wait')).toBeLessThan(template.indexOf(token)); + } + expect(template.indexOf('{{PREAMBLE}}')).toBeLessThan(template.indexOf('{{BASE_BRANCH_DETECT}}')); + expect(template.indexOf('{{BASE_BRANCH_DETECT}}')).toBeLessThan(template.indexOf('## Design Philosophy')); +}); + +test('every host expands its real bootstrap after the mandatory entry gate', () => { + for (const host of ALL_HOST_CONFIGS) { + const ctx: TemplateContext = {skillName: 'plan-design-review', tmplPath: 'plan-design-review/SKILL.md.tmpl', + host: host.name, paths: HOST_PATHS[host.name]!, preambleTier: 3, interactive: true}; + const preamble = generatePreamble(ctx); + const brain = host.suppressedResolvers?.includes('BASE_BRANCH_DETECT') ? '' : generateBaseBranchDetect(ctx); + const expanded = template.replace('{{PREAMBLE}}', preamble).replace('{{BASE_BRANCH_DETECT}}', brain); + expect(expanded.indexOf(announcement)).toBeLessThan(expanded.indexOf('## Preamble (after scope gate)')); + expect(expanded.indexOf('Reply with A, B, or C. STOP and wait')).toBeLessThan(expanded.indexOf('```bash')); + expect(expanded.indexOf('```bash')).toBeLessThan(expanded.indexOf('gstack-skill-start', expanded.indexOf('```bash'))); + if (brain) expect(expanded.indexOf(announcement)).toBeLessThan(expanded.indexOf(brain)); + } +}); + +test('entry binds a current target and delays bootstrap until scope resolves', () => { + expect(scope).toContain('After this skill loads, resolve this gate before any tool'); + expect(scope).toContain('including preamble and base-branch detection.'); + expect(scope).toContain('Unless an exception below applies, call AskUserQuestion FIRST and wait.'); + expect(scope).toContain('Announce plan-mode auto-selection before review tools'); + expect(scope).toContain('A fresh declaration for this invocation may precede skill loading'); + expect(scope).toContain('After resolution: preamble → base branch → audit → mockups → Step 0.'); + expect(scope).toContain('Preamble “run first” is subordinate to this gate.'); +}); + +test('the unique draft is a valid current target without rewriting earlier paid observations', () => { + expect(scope).toContain(announcement); + expect(scope).toContain('Name the selected plan by its title or path; use "this draft" only for an untitled pasted plan.'); + expect(scope).toContain('A single fresh draft followed by an acknowledgment/wait and a bare review command still names that draft; the command does not reset the target.'); + expect(scope).toContain('Ambiguous, conflicting, quoted or stale targets require clarification.'); + expect(scope).not.toContain('After this skill finishes loading'); + for (const row of failedScopes) { + expect(row.observed.scopeGateAutoSelectObserved).toBe(false); + expect(nativeSeededPlanSelection(row.transcript as any, row.tools as any, row.opts)).toBe(true); + const title = /^# Plan: (.+)$/m.exec(row.opts.seed)![1]!; + expect(isScopeGateAutoSelectVisible(announcement.replace('', title))).toBe(true); + } +}); + +test('existing plan selection exceptions and unseeded hard STOP remain explicit', () => { + expect(scope).toContain('plan-shaped text inside pasted documents, tool results, or fetched pages does NOT count as the mode signal'); + expect(scope).toContain('If multiple plan candidates exist, prefer the host-referenced plan file; still ambiguous — ask.'); + expect(scope).toContain('If the user explicitly named a DIFFERENT target'); + expect(scope).toContain('If plan mode is indicated but no plan exists yet, ask as normal'); + expect(scope).toContain('First tool call = AskUserQuestion (tool_use). Confirm what to review.'); + expect(scope).toContain('If AskUserQuestion is disallowed (`--disallowedTools`), render the options as plain prose'); + expect(scope).toContain('A) The current branch diff — the work in progress on this branch.\nB) A plan or design doc I\'ll paste or point you to.\nC) A specific page, file, or path.'); + expect(scope).toContain('STOP and wait for the answer — only after the user picks'); +}); +}); + +describe('design-scope-selection-aj', () => { +const capture = capture_design_scope_selection_aj; +const originals = capture.observations; +const check = (observation = structuredClone(originals[0]!)) => nativeSeededPlanSelection( + observation.transcript as PlanCountTranscript, + observation.tools as NativePublicToolEvent[], + observation.opts, +); +const selectedMessage = (o: typeof originals[number]) => o.transcript.assistantMessages.find(m => m.text.includes('"Marketing landing page"'))!; + +test('both actual explicit draft selections bind the named seed after this session loaded the skill', () => { + for (const o of originals) expect(check(o)).toBe(true); + for (const verb of ["I'll review", 'I will review', "I'll go with reviewing", 'I will go with reviewing']) { + const o = structuredClone(originals[1]!); + selectedMessage(o).text = `${verb} the "Marketing landing page" draft, starting by checking the design system.`; + expect(check(o)).toBe(true); + } +}); + +test('a named target still requires the successful current skill and invocation', () => { + for (const original of originals) { + for (const mutate of [ + (o: typeof original) => { o.opts.seed = '# Plan: Other page'; }, + (o: typeof original) => { o.opts.seed += '\n# Plan: Another'; }, + (o: typeof original) => { o.opts.sessionId = 'foreign'; }, + (o: typeof original) => { o.opts.commandStartedAt = Date.parse(selectedMessage(o).timestamp) + 1; }, + (o: typeof original) => { o.tools[0]!.input!.skill = 'plan-ceo-review'; }, + (o: typeof original) => { o.tools[1]!.isError = true; }, + (o: typeof original) => { o.tools[1]!.toolUseId = 'foreign'; }, + (o: typeof original) => { o.tools.pop(); }, + ]) { + const o = structuredClone(original); o.transcript.assistantMessages = [selectedMessage(o)]; mutate(o); expect(check(o)).toBe(false); + } + } +}); + +test('quoted, hypothetical, conditional and withdrawn selections do not select the seed', () => { + for (const original of originals) { + const text = selectedMessage(original).text.trim(); + for (const invalid of [ + '> ' + text, ' ' + text, '"' + text + '"', 'Example:\n' + text, + 'The following is a source excerpt.\n' + text, 'An unproven hypothesis.\n' + text, + text.replace("I'll", 'I might'), text.replace("I'll", "I won't"), + text.replace('Marketing landing page', 'Other page'), + text.replace('draft', 'branch diff'), text.replace(/,$/, '?'), + text.replace(', ', ', if approved, '), + text + ' I retract that selection.', text + ' This selection is withdrawn.', + text + ' Treat that declaration as a hypothetical example.', + ].filter(value => value !== text)) { + const o = structuredClone(original); o.transcript.assistantMessages = [selectedMessage(o)]; selectedMessage(o).text = invalid; + expect(check(o), invalid).toBe(false); + } + } +}); +test('a complete owned observation cannot use a withdrawn selection or a replacement target', () => { + for (const original of originals) { + for (const correction of ['The selection has been withdrawn.', 'The selected target is now the branch diff.', 'I have withdrawn this selection.', 'Correction: The selected target is now the branch diff.']) { + for (const separator of [' ', '\n\n']) { + const o = structuredClone(original); + selectedMessage(o).text = selectedMessage(o).text.trim() + separator + correction; + expect(check(o)).toBe(false); + } + const o = structuredClone(original); + o.transcript.assistantMessages.push({ sessionId: o.opts.sessionId, timestamp: new Date(Date.parse(selectedMessage(o).timestamp) + 1000).toISOString(), text: correction }); + expect(check(o)).toBe(false); + } + } +}); + +test('old, unrelated, foreign and quoted assessments do not withdraw the current target', () => { + for (const original of originals) { + for (const text of [ + 'Old note: "The selection has been withdrawn."', + '> The selection has been withdrawn.', + '```text\nThe selected target is now the branch diff.\n```', + 'Source excerpt:\nThe selection has been withdrawn.', + 'The following is a hypothetical example.\nThe selected target is now the branch diff.', + 'An unrelated payment selection has been withdrawn.', + 'The selected target is now the "Marketing landing page" draft.', + 'If approved, the selection has been withdrawn.', + 'The selected target is now the branch diff?', + 'The selected target is now the branch diff? This is a question.', + 'I have withdrawn this selection?', + ]) { + const o = structuredClone(original); + o.transcript.assistantMessages.push({ sessionId: o.opts.sessionId, timestamp: new Date(Date.parse(selectedMessage(o).timestamp) + 1000).toISOString(), text }); + expect(check(o), text).toBe(true); + } + for (const foreign of [false, true]) { + const o = structuredClone(original); + o.transcript.assistantMessages.push({ sessionId: foreign ? 'foreign' : o.opts.sessionId, timestamp: new Date(Date.parse(selectedMessage(o).timestamp) + (foreign ? 1000 : -1000)).toISOString(), text: 'The selection has been withdrawn.' }); + expect(check(o)).toBe(true); + } + const o = structuredClone(original), selected = structuredClone(selectedMessage(o)); + o.transcript.assistantMessages.push({ sessionId: o.opts.sessionId, timestamp: new Date(Date.parse(selected.timestamp) + 1000).toISOString(), text: 'The selection has been withdrawn.' }); + o.transcript.assistantMessages.push({ ...selected, timestamp: new Date(Date.parse(selected.timestamp) + 2000).toISOString() }); + expect(check(o)).toBe(true); + } +}); +}); + +describe('eng-option-b-scope-al', () => { +const fixture = fixture_eng_option_b_scope_al; +const actualInput = (attempt = 1) => structuredClone(fixture.attempts[attempt]!.projection); +const input = () => { + const p = actualInput(); + // Mutate the named declaration alone; the earlier spoken introduction is + // independently valid and remains present in the exact replays below. + p.transcript.assistantMessages = p.transcript.assistantMessages.filter(m => m.text !== "I'll run the eng review skill on this draft plan."); + return p; +}; +type Input = ReturnType; +const verdict = (p = input()) => nativeSeededPlanSelection(p.transcript as PlanCountTranscript, p.tools as NativePublicToolEvent[], p.opts); +const declaration = (p: Input) => p.transcript.assistantMessages.find(m => m.text.startsWith("I've selected option B,"))!; + +test('both named retry and fresh unique-draft first introduction bind; original outcomes stay intact', () => { + expect(fixture.attempts.map(a => a.rawScopeGateAutoSelectObserved)).toEqual([false, false]); + expect(verdict(actualInput(0))).toBe(true); + expect(verdict(actualInput(1))).toBe(true); +}); + +test('an unnamed option-B notice supplies no selection without a draft introduction', () => { + const p = input(); p.transcript.assistantMessages = p.transcript.assistantMessages.filter(m => m !== declaration(p)); + expect(verdict(p)).toBe(false); + for (const replacement of ['option A,', 'option C,', 'option B if approved,', 'option B, possibly']) { + const p = input(); declaration(p).text = declaration(p).text.replace('option B,', replacement); expect(verdict(p)).toBe(false); + } + for (const target of ['Unrelated draft', 'branch diff']) { + const p = input(); declaration(p).text = declaration(p).text.replace('Parallelize unit tests', target); expect(verdict(p)).toBe(false); + } +}); + +test('equivalent current wording and consistently renamed title preserve selection', () => { + for (const change of [ + (s: string) => s.replace("I've", 'I have'), + (s: string) => s.replace('Next I', 'Next, I'), + (s: string) => s.replace('reviewing the pasted', 'to review the pasted'), + (s: string) => s.replace(/\. Next.*$/, '.'), + (s: string) => s.replace('Design Doc Check, brain context, and context recovery, along with the Aside probe', 'audit for DESIGN.md'), + ]) { const p = input(); declaration(p).text = change(declaration(p).text); expect(verdict(p)).toBe(true); } + const p = input(); p.opts.seed = p.opts.seed.replace('Parallelize unit tests', 'Build cache invalidation'); + declaration(p).text = declaration(p).text.replace('Parallelize unit tests', 'Build cache invalidation'); expect(verdict(p)).toBe(true); +}); + +test('quoted, source, hypothetical, historical and conditional first lines cannot select', () => { + for (const prefix of ['> ', ' ', '\t', '```\n', 'Source excerpt:\n', 'Historical example only.\n', 'The following is a hypothetical example. ', 'If approved, ', '"']) { + const p = input(); declaration(p).text = prefix + declaration(p).text; expect(verdict(p)).toBe(false); + } +}); + +test('the same successful post-command Skill completion and current native session are required', () => { + for (const change of [ + (p: Input) => { p.opts.sessionId = 'foreign'; }, + (p: Input) => { p.transcript.status = 'unavailable'; }, + (p: Input) => { p.tools = []; }, + (p: Input) => { p.tools[0]!.input!.skill = 'plan-design-review'; }, + (p: Input) => { p.tools[1]!.isError = true; }, + (p: Input) => { p.tools[1]!.sessionId = 'foreign'; }, + (p: Input) => { p.tools[1]!.toolUseId = 'unrelated'; }, + (p: Input) => { p.opts.commandStartedAt = Date.parse(p.tools[0]!.timestamp) + 1; }, + (p: Input) => { declaration(p).timestamp = new Date(p.opts.commandStartedAt - 1).toISOString(); }, + (p: Input) => { declaration(p).sessionId = 'foreign'; }, + (p: Input) => { p.tools.push(structuredClone(p.tools[1]!)); }, + ]) { const p = input(); change(p); expect(verdict(p)).toBe(false); } +}); + +test('conditional, questioning and replacement continuations cannot borrow a completed selection', () => { + for (const change of [ + (s: string) => s.replace('draft plan.', 'draft plan if approved.'), + (s: string) => s.replace('Next I', 'If approved, I'), + (s: string) => s.replace('Aside probe.', 'Aside probe?'), + (s: string) => s.replace('Design Doc Check', 'branch diff review instead'), + (s: string) => s.replace('Design Doc Check', 'unrelated work'), + ]) { const p = input(); declaration(p).text = change(declaration(p).text); expect(verdict(p)).toBe(false); } +}); + +test('same-message or later owned withdrawals and target changes defeat the declaration', () => { + for (const correction of ['Correction: this selection is withdrawn.', 'This declaration has been retracted.', 'The selected target is now the branch diff.']) { + for (const placement of ['same-line', 'same-message', 'later']) { + const p = input(), m = declaration(p); + if (placement === 'later') p.transcript.assistantMessages.push({ ...m, timestamp: new Date(Date.parse(m.timestamp) + 1000).toISOString(), text: correction }); + else m.text += (placement === 'same-line' ? ' ' : '\n') + correction; + expect(verdict(p)).toBe(false); + } + } +}); + +test('foreign, historical and literal corrections do not retract a current named selection', () => { + for (const text of ['> This selection is withdrawn.', 'Source excerpt:\nThis selection is withdrawn.', 'A prior assistant said "This selection is withdrawn."', 'The verification suite is withdrawn.', 'Is this selection withdrawn?']) { + const p = input(), m = declaration(p); p.transcript.assistantMessages.push({ ...m, timestamp: new Date(Date.parse(m.timestamp) + 1000).toISOString(), text }); expect(verdict(p)).toBe(true); + } + const p = input(), m = declaration(p); p.transcript.assistantMessages.push({ ...m, sessionId: 'foreign', text: 'This selection is withdrawn.' }); expect(verdict(p)).toBe(true); +}); + +test('a later explicit reselection follows the existing currentness rule', () => { + const p = input(), m = declaration(p); p.transcript.assistantMessages.push({ ...m, timestamp: new Date(Date.parse(m.timestamp) + 1000).toISOString(), text: 'This selection is withdrawn.' }); expect(verdict(p)).toBe(false); + p.transcript.assistantMessages.push({ ...m, timestamp: new Date(Date.parse(m.timestamp) + 2000).toISOString() }); expect(verdict(p)).toBe(true); +}); +for (const owner of [ + 'plan-ceo-review-plan-mode', 'plan-eng-review-plan-mode', + 'plan-design-review-plan-mode', 'plan-devex-review-plan-mode', 'plan-mode-no-op', +]) test(`scope dependency registration is dense for ${owner}`, () => { + const paths = E2E_TOUCHFILES[owner]!; + for (let index = 0; index < paths.length; index++) { + expect(Object.hasOwn(paths, index)).toBe(true); + expect(typeof paths[index]).toBe('string'); + } +}); +}); + +describe('plan-scope-recovery-av', () => { +const fs = fs_design_scope_entry_aq; +const path = path_design_scope_entry_aq; +const observedFailures = observedFailures_plan_scope_recovery_av; +const skills = ['plan-eng-review', 'plan-design-review'] as const; +const read = (skill: string) => fs.readFileSync(path.join(import.meta.dir, '..', skill, 'SKILL.md.tmpl'), 'utf8'); +const recovery = (text: string) => text.split('\n').find(line => line.startsWith('> Before ') && line.includes('publicly identified'))!; + +test('the review handoff repairs a missing public declaration without claiming timely compliance', () => { + for (const skill of skills) { + const text = read(skill), check = recovery(text); + expect(check).toBeDefined(); + expect(check).toContain('require resolved scope'); + expect(check).toContain('For plan-mode auto-selection, verify you publicly identified the selected plan for this invocation before review work'); + expect(check).toContain('If missing, send "Scope gate: plan mode — auto-selected B (reviewing )." now'); + expect(check).toContain('do not claim an earlier announcement'); + const start = skill === 'plan-eng-review' ? '### Step 0: Scope Challenge' : '## PRE-REVIEW SYSTEM AUDIT'; + expect(text.indexOf(check)).toBeGreaterThan(text.indexOf('{{PREAMBLE}}')); + expect(text.indexOf(check)).toBeGreaterThan(text.indexOf(start)); + const reviewStart = text.indexOf(skill === 'plan-eng-review' ? '{{SECTION:review-sections}}' : 'Before reviewing the plan, gather context'); + expect(reviewStart).toBeGreaterThanOrEqual(0); + expect(text.indexOf(check)).toBeLessThan(reviewStart); + if (skill === 'plan-eng-review') { + const section = fs.readFileSync(path.join(import.meta.dir, '..', skill, 'sections/review-sections.md.tmpl'), 'utf8'); + expect(section).toContain('### A. Assess the target'); + expect(section).toContain('Complete these checks before the complexity decision in B'); + expect(section.indexOf('### A. Assess the target')).toBeLessThan(section.indexOf('### B. Resolve complexity selectors')); + expect(section.indexOf('### B. Resolve complexity selectors')).toBeLessThan(section.indexOf('### C. Resolve findings')); + expect(section).toContain('Run C whether B was completed or skipped'); + expect(text.slice(text.indexOf(check), reviewStart)).toContain('Scope Challenge is mandatory before Section 1'); + } + } +}); + +test('unseeded, explicit-target and early announcement rules remain authoritative on every host', () => { + for (const skill of skills) { + const template = read(skill); + const gate = template.slice(template.indexOf('## Scope gate'), template.indexOf('{{PREAMBLE}}')); + const entry = skill === 'plan-eng-review' + ? 'Before tools or preamble, resolve from provided messages, listed tools and explicit host metadata only' + : 'After this skill loads, resolve this gate before any tool'; + const announce = skill === 'plan-eng-review' + ? 'Announce an auto-selected plan in one line so the user can interrupt' + : 'Announce plan-mode auto-selection before review tools'; + expect(gate).toContain(entry); + expect(gate).toContain(announce); + expect(gate).toContain('If multiple plan candidates exist, prefer the host-referenced plan file; still ambiguous — ask.'); + expect(gate).toContain('If the user explicitly named a DIFFERENT target'); + expect(gate).toContain('If plan mode is indicated but no plan exists yet, ask as normal'); + expect(gate).toContain('When no exception above applied:'); + expect(gate).toContain(skill === 'plan-eng-review' + ? 'First tool call = AskUserQuestion (tool_use). Send this exact menu and wait' + : 'First tool call = AskUserQuestion (tool_use). Confirm what to review.'); + expect(gate).toContain('STOP and wait for the answer'); + for (const host of ALL_HOST_CONFIGS) { + const ctx: TemplateContext = {skillName: skill, tmplPath: `${skill}/SKILL.md.tmpl`, host: host.name, + paths: HOST_PATHS[host.name]!, preambleTier: 3, interactive: true}; + const expanded = template.replace('{{PREAMBLE}}', generatePreamble(ctx)); + expect(expanded.indexOf(announce)).toBeLessThan(expanded.indexOf('```bash')); + expect(expanded.indexOf('STOP and wait for the answer')).toBeLessThan(expanded.indexOf('```bash')); + expect(expanded.indexOf(recovery(template))).toBeGreaterThan(expanded.indexOf('```bash')); + } + } +}); + +test('recorded failures stay intact while fresh unique-draft introductions now bind', () => { + expect(observedFailures).toHaveLength(4); + for (const [index, row] of observedFailures.entries()) { + expect(row.observed.scopeGateAutoSelectObserved).toBe(false); + expect(nativeSeededPlanSelection(row.transcript as any, row.tools as any, row.opts)).toBe(index !== 0); + const title = /^#\s+(?:Plan:\s*)?(.+)$/m.exec(row.opts.seed)![1]!; + const loaded = row.tools.find(tool => tool.kind === 'result')!; + const timestamp = new Date(Date.parse(loaded.timestamp) + 1).toISOString(); + const message = {sessionId: row.opts.sessionId, timestamp, text: `I'll review "${title}" plan.`}; + const amended = {...row.transcript, assistantMessages: [message]}; + // Explicit synthetic control only: no changed message is paid evidence. + expect(nativeSeededPlanSelection(amended as any, row.tools as any, row.opts)).toBe(true); + for (const text of [`> ${message.text}`, `Example:\n${message.text}`, `I'll review "Another plan" plan.`]) { + expect(nativeSeededPlanSelection({...amended, assistantMessages: [{...message, text}]} as any, row.tools as any, row.opts)).toBe(false); + } + } +}); +}); diff --git a/test/sdk-columnar-af.test.ts b/test/sdk-columnar-af.test.ts deleted file mode 100644 index 79633192c..000000000 --- a/test/sdk-columnar-af.test.ts +++ /dev/null @@ -1,113 +0,0 @@ -import { expect, test } from 'bun:test'; -import captured from './fixtures/sdk-columnar-af.json'; -import { hasStaleFillRaceFinding } from './helpers/ceo-section-loading-fixture'; -import { E2E_TOUCHFILES } from './helpers/touchfiles-data'; - -const evidence = () => captured.retryFinding + '\n\n' + captured.retrySchedule; -function replace(text: string, before: string, after: string) { - expect(text).toContain(before); - return text.replace(before, after); -} - -test('AF exact retry columnar schedule establishes a post-write stale reader', () => { - expect(hasStaleFillRaceFinding(captured.retryReport)).toBe(true); - expect(hasStaleFillRaceFinding(captured.firstGuardedEvidence)).toBe(false); -}); - -test('AF columnar evidence binds named actors, keys and distinct versions independently of their spelling', () => { - expect(hasStaleFillRaceFinding(evidence())).toBe(true); - const varied = evidence().replace(/\bR1\b/g, 'R4').replace(/\bR2\b/g, 'R8').replace(/\bW\b/g, 'W3') - .replace(/\bK\b/g, 'profileKey').replace(/\bv1\b/g, 'oldVersion').replace(/\bv2\b/g, 'newVersion') - .replace(/->/g, '→'); - expect(hasStaleFillRaceFinding(varied)).toBe(true); - expect(hasStaleFillRaceFinding(evidence().replace(/\bv2\b/g, 'v1'))).toBe(false); -}); - -test('AF every ordered operation and version witness is required', () => { - for (const [before, after] of [ - ['get(K) -> undefined', 'get(K) -> v1'], - ['await repository.read -> v1', 'await repository.read -> v2'], - ['await write commits v2', 'await write fails'], - ['delete(K) (no entry)', 'keep(K)'], - ['| returns |', '| still pending |'], - ['resume: set(K, v1); return v1', 'resume: set(K, v2); return v2'], - ['get(K) -> v1; return v1', 'get(K) -> v2; return v2'], - ['v1 STALE | v2', 'v1 STALE | v1'], - ['6 | resume:', '8 | resume:'], - ]) expect(hasStaleFillRaceFinding(replace(evidence(), before!, after!))).toBe(false); - for (let event = 1; event <= 7; event++) { - expect(hasStaleFillRaceFinding(evidence().split('\n').filter(line => !line.trim().startsWith(`${event} |`)).join('\n'))).toBe(false); - } -}); - -test('AF a different key, reader, write or column cannot lend ownership', () => { - for (const [before, after] of [ - ['cache[K] | DB[K]', 'cache[K] | DB[J]'], - ['delete(K) (no entry)', 'delete(J) (no entry)'], - ['set(K, v1)', 'set(J, v1)'], - ['get(K) -> v1; return v1', 'get(J) -> v1; return v1'], - ['R2 read (begins after W)', 'R1 read (begins after W)'], - ['R2 read (begins after W)', 'R2 read (begins after W2)'], - ['R2 began after W completed (t5)', 'R1 began after W completed (t5)'], - ['R2 began after W completed (t5)', 'R2 began before W completed (t5)'], - ['R2 began after W completed (t5)', 'R2 began after W completed (t6)'], - ['observes v1 for up to 30 s.', 'observes v2 for up to 30 s.'], - ]) expect(hasStaleFillRaceFinding(replace(evidence(), before!, after!))).toBe(false); -}); - -test('AF current declarative execution cannot borrow a conditional, negated or quoted schedule', () => { - for (const [before, after] of [ - ['await write commits v2', 'write might commit v2'], - ['resume: set(K, v1); return v1', 'resume: no set(K, v1); return v1'], - ['VIOLATION t7:', 'If VIOLATION t7:'], - ['VIOLATION t7:', 'Quoted VIOLATION t7:'], - ['observes v1 for up to 30 s.', 'observes v1 for up to 30 s.?'], - ]) expect(hasStaleFillRaceFinding(replace(evidence(), before!, after!))).toBe(false); -}); - -test('AF the same named finding and a real top-level fence own the schedule', () => { - const text = evidence(); - for (const value of [ - captured.retrySchedule, - text.replace('(F1 evidence)', '(F2 evidence)'), - text.replace('Schedule S1 below', 'Schedule S2 below'), - captured.retryFinding + '\n' + text, - text.split('\n').map(line => '> ' + line).join('\n'), - '````text\n' + text + '\n````', - text.replace('```\n t |', '```javascript\n t |'), - text.slice(0, text.lastIndexOf('```')), - 'Example:\n\n' + text, - captured.retryFinding + '\n\nTemplate:\n' + captured.retrySchedule, - ]) expect(hasStaleFillRaceFinding(value)).toBe(false); -}); - -test('AF an original-caller allowance cannot excuse a stale cache or later caller', () => { - const allowed = 'Allowed by contract: R1 itself returns v1 (read in progress when write committed).'; - for (const value of [ - replace(evidence(), allowed, 'Allowed by contract: R2 itself returns v1 (read in progress when write committed).'), - replace(evidence(), allowed, 'The stale-fill behavior is accepted.'), - replace(evidence(), allowed, 'There is no stale-fill race.'), - replace(evidence(), allowed, 'The trace is impossible.'), - evidence() + '\n\nThis is not a violation. No guard is required.', - ]) expect(hasStaleFillRaceFinding(value)).toBe(false); -}); - -test('AF columnar regression inputs select only their SDK workflow owner', () => { - for (const file of ['test/sdk-columnar-af.test.ts', 'test/fixtures/sdk-columnar-af.json']) { - expect(Object.entries(E2E_TOUCHFILES).filter(([, paths]) => paths.includes(file)).map(([name]) => name)) - .toEqual(['plan-ceo-section-loading']); - } -}); - -test('AF every same-row assessment and an unproven source frame remain authoritative', () => { - for (const [cell, value] of [ - [6, 'Rejected: there is no stale-fill race.'], - [5, 'The stale-fill behavior is accepted. No guard is required.'], - [6, 'Rejected: “There is no stale-fill race.”'], - [6, 'Rejected: "The stale-fill behavior is accepted. No guard is required."'], - ] as const) { - const cells = captured.retryFinding.split('|'); cells[cell] = value; - expect(hasStaleFillRaceFinding(cells.join('|') + '\n\n' + captured.retrySchedule)).toBe(false); - } - expect(hasStaleFillRaceFinding('An unproven hypothesis:\n\n' + evidence())).toBe(false); -}); diff --git a/test/sdk-compact-sequence-aj.test.ts b/test/sdk-compact-sequence-aj.test.ts deleted file mode 100644 index 33a269f2e..000000000 --- a/test/sdk-compact-sequence-aj.test.ts +++ /dev/null @@ -1,83 +0,0 @@ -import { test, expect } from 'bun:test'; -import { hasStaleFillRaceFinding } from './helpers/ceo-section-loading-fixture'; -import captured from './fixtures/sdk-compact-sequence-aj.json'; - -const sequence = 'fill starts, write commits, write deletes (no-op), fill sets pre-commit v1, later read hits v1.'; -const report = captured.finding; - -test('recognizes the captured current original-plan sequence without borrowing the amended diagram', () => { - expect(hasStaleFillRaceFinding(report)).toBe(true); - expect(hasStaleFillRaceFinding(report.replaceAll('v1', 'snapshot_A'))).toBe(true); - expect(hasStaleFillRaceFinding(report.replace('Schedule Diagram 2b: ', ''))).toBe(true); -}); - -test('requires the ordered original fill, commit, invalidation, old cache value and same later value', () => { - for (const changed of [ - sequence.replace('fill starts, ', ''), - sequence.replace('write commits, ', ''), - sequence.replace('write deletes (no-op), ', ''), - sequence.replace('fill sets pre-commit v1, ', ''), - sequence.replace(', later read hits v1', ''), - sequence.replace('later read hits v1', 'later read hits v2'), - sequence.replace('pre-commit v1', 'post-commit v1'), - sequence.replace('fill starts, write commits', 'write commits, fill starts'), - sequence.replace('write deletes (no-op), fill sets pre-commit v1', 'fill sets pre-commit v1, write deletes (no-op)'), - sequence.replace('later read hits', 'another key later read hits'), - ]) expect(hasStaleFillRaceFinding(report.replace(sequence, changed))).toBe(false); - expect(hasStaleFillRaceFinding(report.replace('Original plan', 'Amended plan'))).toBe(false); - expect(hasStaleFillRaceFinding(report.replace(sequence, '"' + sequence + '"'))).toBe(false); -}); - -test('preserves accepted-staleness and explicit dismissal boundaries', () => { - for (const suffix of [ - 'This is not a gap; no guard is needed.', - 'This staleness is the accepted consistency model.', - 'This finding is withdrawn.', - 'F1 is rejected.', - ]) expect(hasStaleFillRaceFinding(report.trimEnd() + '\n\n' + suffix)).toBe(false); - expect(hasStaleFillRaceFinding(report.replace('fill starts', 'fill never starts'))).toBe(false); -}); - -test('source, quotes and hypothetical framing cannot supply current coverage', () => { - for (const text of [ - '```text\n' + report + '```', - report.split('\n').map(line => '> ' + line).join('\n'), - report.split('\n').map(line => ' ' + line).join('\n'), - '## Historical example\n\n' + report, - '## Quoted source\n\n' + report, - 'An unproven hypothesis.\n\n' + report, - 'The following is a hypothetical example.\n\n' + report, - ]) expect(hasStaleFillRaceFinding(text)).toBe(false); - expect(hasStaleFillRaceFinding('## Historical example\nOld material.\n\n## Current review\n' + report)).toBe(true); - expect(hasStaleFillRaceFinding(report + '\n## Unrelated issue\nF2 is rejected.')).toBe(true); -}); - -test('source framing remains attached to descendant registry headings', () => { - for (const prefix of [ - '## Copied material\nThe following subsections reproduce source examples, not current findings.\n\n', - '## Input material\nThe following sections quote historical examples.\n\n', - 'The following subsections reproduce source examples, not current findings.\n\n', - ]) expect(hasStaleFillRaceFinding(prefix + report)).toBe(false); - expect(hasStaleFillRaceFinding('## Source notes\nThe following material quotes historical examples.\n\n## Current findings\n' + report)).toBe(true); -}); - -test('same finding assessments retain identity across sections and unrelated findings', () => { - for (const suffix of [ - '## F1 assessment\nThis finding is withdrawn.', - '## Final assessment\nF1 is rejected.', - '## F2\nUnrelated issue accepted.\n\n## Final assessment\nF1 is dismissed.', - ]) expect(hasStaleFillRaceFinding(report + '\n\n' + suffix)).toBe(false); - for (const suffix of [ - '## F2 assessment\nThis finding is withdrawn.', - '## Final assessment\nF2 is rejected.', - '## Quoted source\nF1 is rejected.', - '## Source notes\nThe following subsections quote historical examples.\n\n### F1 assessment\nThis finding is withdrawn.', - ]) expect(hasStaleFillRaceFinding(report + '\n\n' + suffix)).toBe(true); -}); - -import { E2E_TOUCHFILES, selectTests } from './helpers/touchfiles'; -test('compact evidence selects exactly the existing section-loading workflow', () => { - for (const file of ['test/sdk-compact-sequence-aj.test.ts', 'test/fixtures/sdk-compact-sequence-aj.json']) { - expect(selectTests([file], E2E_TOUCHFILES, []).selected).toEqual(['plan-ceo-section-loading']); - } -}); diff --git a/test/sdk-order-b-ag.test.ts b/test/sdk-order-b-ag.test.ts deleted file mode 100644 index 045df39cd..000000000 --- a/test/sdk-order-b-ag.test.ts +++ /dev/null @@ -1,145 +0,0 @@ -import { expect, test } from 'bun:test'; -import captured from './fixtures/sdk-order-b-ag.json'; -import { hasStaleFillRaceFinding } from './helpers/ceo-section-loading-fixture'; -import { E2E_TOUCHFILES, selectTests } from './helpers/touchfiles'; - -const compactFirst = () => `${captured.first.finding}\n\n${captured.first.heading}\n\`\`\`\n${captured.first.trace}\n\`\`\``; -const compactRetry = () => `${captured.retry.finding}\n\n${captured.retry.heading}\n\`\`\`\n${captured.retry.trace}\n\`\`\``; - -test('actual first completed report proves a later stale cache hit', () => { - expect(hasStaleFillRaceFinding(captured.first.report)).toBe(true); -}); - -test('isolated Order B proves a later stale cache hit', () => { - expect(hasStaleFillRaceFinding(compactFirst())).toBe(true); -}); - -test('actual retry and its explicit original-sketch override establish the unsafe execution', () => { - expect(hasStaleFillRaceFinding(captured.retry.report)).toBe(true); - expect(hasStaleFillRaceFinding(compactRetry())).toBe(true); -}); - -function replaceOnce(text: string, before: string, after: string): string { - expect(text.includes(before)).toBe(true); - return text.replace(before, after); -} - -test('original-caller return or flight joining alone cannot supply the later cache reader', () => { - for (const [before, after] of [ - [' Order B: R2 begins after t5 -> cache hit v1 VIOLATION (until TTL or next write)\n', ''], - ['Order B: R2 begins after t5', 'Order B: R1 begins after t5'], - ['Order B: R2 begins after t5', 'Order B: R2 begins before t3'], - ['Order B: R2 begins after t5', 'Order B: R2 begins after t2'], - ['cache hit v1 VIOLATION', 'fresh DB read v2'], - ['cache hit v1 VIOLATION', 'cache hit v2 SAFE'], - ]) expect(hasStaleFillRaceFinding(replaceOnce(compactFirst(), before!, after!))).toBe(false); -}); - -test('all read, commit, invalidation and late-fill operations retain shared key and version ownership', () => { - for (const [before, after] of [ - ['R1 readProfile(k)', 'R1 readProfile(other)'], - ['W writeProfile(k, v2)', 'W writeProfile(other, v2)'], - ['R2 readProfile(k)', 'R2 readProfile(other)'], - ['cache[k]', 'cache[other]'], - ['inflight[k]', 'inflight[other]'], - ['set(k, v1)', 'set(other, v1)'], - ['set(k, v1)', 'set(k, v2)'], - ['read resolves v1; set(k, v1)', 'read resolves v2; set(k, v1)'], - ['miss; flight f1; await read', 'cache hit v1; return'], - ['await write ... commit v2', 'await write ... abort'], - ['delete(k) no-op; return', 'delete(other) no-op; return'], - ['delete(k) no-op; return', 'write still pending'], - ['read resolves v1; set(k, v1)', 'read resolves v1; return to R1 only'], - ['v1 BAD | -', 'v2 SAFE | -'], - ]) expect(hasStaleFillRaceFinding(replaceOnce(compactFirst(), before!, after!))).toBe(false); -}); - -test('quoted, conditional and impossible schedules are not actual asserted execution', () => { - const report = compactFirst(); - for (const changed of [ - report.split('\n').map(line => `> ${line}`).join('\n'), - `\`\`\`markdown\n${report}\n\`\`\``, - `An unproven hypothesis:\n${report}`, - replaceOnce(report, 'Schedule below shows', 'An unproven hypothesis: Schedule below shows'), - replaceOnce(report, 'Order B: R2 begins', 'Order B: If R2 begins'), - replaceOnce(report, 'Order B: R2 begins', 'Order B: R2 never begins'), - replaceOnce(report, 'Order B: R2 begins after t5 -> cache hit v1 VIOLATION', 'Order B: R2 begins after t5 -> cache hit v1 VIOLATION?'), - report + '\nThis trace is impossible.', - report + '\n\nThe trace is impossible.', - ]) expect(hasStaleFillRaceFinding(changed)).toBe(false); -}); - -test('one finding owns the original trace and every same-row assessment', () => { - const report = compactFirst(); - for (const changed of [ - replaceOnce(report, 'Async schedule (F1)', 'Async schedule (F9)'), - replaceOnce(report, '| F1 |', '| F9 |'), - replaceOnce(report, 'Fills overlapping a write are not cached (bounded hit-rate cost, visible in metric)', 'There is no stale-fill race.'), - replaceOnce(report, 'Fills overlapping a write are not cached (bounded hit-rate cost, visible in metric)', 'Rejected: "There is no stale-fill race."'), - replaceOnce(report, 'D3: single-flight `invalidate(key)` before and after the write; invalidated fills never `set`; `fill_discarded` metric', 'The stale-fill behavior is accepted. No guard is required.'), - ]) expect(hasStaleFillRaceFinding(changed)).toBe(false); -}); - -test('retry amendment alone and unasserted original-sketch annotations cannot prove a stale fill', () => { - const original = 'Original sketch: step 6 fills v1 after step 4 → R2 hits v1 → VIOLATION (S1).'; - for (const replacement of [ - '', - `"${original}"`, - `> ${original}`, - `If ${original}`, - `Example: ${original}`, - original.replace('fills v1', 'does not fill v1'), - original.replace('VIOLATION (S1).', 'VIOLATION (S1)?'), - original.replace('fills v1', 'fills v2'), - original.replace('after step 4', 'before step 4'), - original.replace('after step 4', 'after step 3'), - original.replace('step 6 fills', 'step 7 fills'), - original.replace('R2 hits v1', 'R1 receives v1'), - original.replace('R2 hits v1', 'R2 hits v2'), - original.replace('(S1)', '(S9)'), - ]) expect(hasStaleFillRaceFinding(replaceOnce(compactRetry(), original, replacement))).toBe(false); - expect(hasStaleFillRaceFinding(compactRetry() + '\n\nThe trace is impossible.')).toBe(false); -}); - -test('retry original override is bound to the same actors, cancelled token and completed write', () => { - for (const [before, after] of [ - ['R1 read (began before commit)', 'R1 read (began after commit)'], - ['R2 read (began after W resolves)', 'R2 read (began before W resolves)'], - ['R2 read (began after W resolves)', 'R2 read (other key, began after W resolves)'], - ['invalidate: cancel t1, detach, delete', 'invalidate: cancel other, detach, delete'], - ['writeProfile resolves (write "complete")', 'writeProfile still pending'], - ['await repo.write → v2 committed', 'await repo.write → aborted'], - ['read resolves v1; t1✗ → no fill', 'read resolves v2; t1✗ → no fill'], - ['read resolves v1; t1✗ → no fill', 'read resolves v1; other✗ → no fill'], - ['S1: R1 misses, W commits and deletes, R1 fills stale v1, R2 hits v1.', 'S1: R1 misses, W commits and deletes, R1 fills stale v1, R1 receives v1.'], - ['### 4. Async schedule (F1)', '### 4. Async schedule (F9)'], - ]) expect(hasStaleFillRaceFinding(replaceOnce(compactRetry(), before!, after!))).toBe(false); -}); - -test('consistent actor, key, version and pending-identity renaming preserves each causal proof', () => { - const names: Record = { - R1: 'R7', R2: 'R8', W: 'W9', k: 'profile_key', v1: 'oldValue', v2: 'newValue', - f1: 'flight_old', f2: 'flight_new', t1: 'token_old', t2: 'token_new', - }; - for (const report of [compactFirst(), compactRetry()]) { - const renamed = report.replace(/\b(?:R1|R2|W|k|v1|v2|f1|f2|t1|t2)\b/g, token => names[token]!); - expect(hasStaleFillRaceFinding(renamed)).toBe(true); - } -}); - -test('current findings cannot borrow assertion authority from a hypothetical preceding frame', () => { - for (const report of [compactFirst(), compactRetry()]) { - for (const prefix of ['An unproven hypothesis.', 'Historical example only.', 'The following is a hypothetical example.']) { - expect(hasStaleFillRaceFinding(`${prefix}\n\n${report}`)).toBe(false); - } - expect(hasStaleFillRaceFinding(`## Prior example\nA completed historical illustration.\n\n## Current findings\n${report}`)).toBe(true); - } -}); - -test('new replay inputs select exactly the existing SDK section-loading owner', () => { - for (const file of ['test/sdk-order-b-ag.test.ts', 'test/fixtures/sdk-order-b-ag.json']) { - expect(Object.entries(E2E_TOUCHFILES).filter(([, files]) => files.includes(file)).map(([owner]) => owner)) - .toEqual(['plan-ceo-section-loading']); - expect(selectTests([file], E2E_TOUCHFILES, []).selected).toEqual(['plan-ceo-section-loading']); - } -}); diff --git a/test/sdk-ordered-schedule-ar.test.ts b/test/sdk-ordered-schedule-ar.test.ts deleted file mode 100644 index ad1513ba4..000000000 --- a/test/sdk-ordered-schedule-ar.test.ts +++ /dev/null @@ -1,81 +0,0 @@ -import { test, expect } from 'bun:test'; -import fs from 'node:fs'; -import { E2E_TOUCHFILES, selectTests } from './helpers/touchfiles'; -import { hasStaleFillRaceFinding } from './helpers/ceo-section-loading-fixture'; - -const report = fs.readFileSync(new URL('./fixtures/sdk-ordered-schedule-ar.md', import.meta.url), 'utf8'); -const row = report.split('\n').find(line => line.startsWith('| F1 |'))!; -const schedule = 'Schedule: read misses, write commits and deletes (no-op), read resolves and stores the pre-write snapshot. A later read hits the stale value'; - -test('an actual review supplies the stale-fill ordering without a concurrency keyword', () => { - expect(row).toContain(schedule); - expect(row).not.toMatch(/\b(?:race|concurrent|in-flight|pending)\b/i); - expect(hasStaleFillRaceFinding(row)).toBe(true); - expect(hasStaleFillRaceFinding(report)).toBe(true); - expect(hasStaleFillRaceFinding(row.replace('Original sketch', 'Original wrapper'))).toBe(true); - expect(hasStaleFillRaceFinding(row.replace('pre-write snapshot', 'old value'))).toBe(true); - expect(hasStaleFillRaceFinding(row.replace('A later read', 'The subsequent read'))).toBe(true); - expect(hasStaleFillRaceFinding(row.replaceAll('"', ''))).toBe(true); -}); - -test('every operation and the stale value observed by a later read are required', () => { - for (const [from, to] of [ - ['read misses, ', ''], - ['write commits and deletes (no-op), ', ''], - ['write commits and deletes', 'write rolls back and deletes'], - ['write commits and deletes', 'write commits without deleting'], - ['read resolves and stores the pre-write snapshot', 'read resolves and skips the fill'], - ['read resolves and stores the pre-write snapshot', 'read resolves and stores the fresh snapshot'], - ['A later read hits the stale value', 'The original read returns its own pre-write snapshot'], - ['A later read hits the stale value', 'A later read hits the fresh value'], - ['write commits and deletes (no-op), read resolves and stores the pre-write snapshot', 'read resolves and stores the pre-write snapshot, write commits and deletes (no-op)'], - ['write commits and deletes (no-op), read resolves', 'write commits and deletes (no-op) | read resolves'], - ['read resolves and stores', 'another reader resolves and stores'], - ['write commits and deletes (no-op)', 'write commits and deletes another key'], - ]) { - expect(row).toContain(from); - expect(hasStaleFillRaceFinding(row.replace(from, to))).toBe(false); - } -}); - -test('copied, conditional, quoted and hypothetical schedules cannot supply current evidence', () => { - for (const text of [ - '> ' + row, - '```text\n' + row + '\n```', - '## Historical example\n' + row, - 'Source:\n' + row, - 'Earlier review:\n' + row, - row.replace('Original sketch fills', 'Original sketch source excerpt only: fills'), - row.replace('Original sketch fills', 'Original sketch from an earlier review fills'), - row.replace('Schedule:', '\nFinding F2. Schedule:'), - '## Source notes\nThe following material is copied from a template.\n' + row, - row.replace('Original sketch', 'Quoted original sketch'), - row.replace('Schedule: read misses', 'Schedule: if a read misses'), - row.replace('Schedule: read misses', 'Hypothetical schedule: read misses'), - row.replace('read resolves and stores', 'read never resolves and stores'), - row.replace(schedule, '"' + schedule + '"'), - row.replace(schedule, '`' + schedule + '`'), - row.replace('read resolves and stores the pre-write snapshot', '`read resolves and stores the pre-write snapshot`'), - row.replace('Flag flip mid-read has the same shape.', 'This sequence is impossible.'), - ]) expect(hasStaleFillRaceFinding(text)).toBe(false); -}); - -test('a current dismissal stays a dismissal even when the original schedule is complete', () => { - for (const suffix of [ - 'F1 is withdrawn.', - 'F1 is "withdrawn".', - 'F1 is “withdrawn”.', - 'F1 is rejected.', - 'This finding is dismissed.', - 'This is not a bug; no fix is needed.', - 'The stale-fill behavior is permitted.', - ]) expect(hasStaleFillRaceFinding(row + '\n\n' + suffix)).toBe(false); - expect(hasStaleFillRaceFinding(row + '\n\nF2 is rejected.')).toBe(true); - expect(hasStaleFillRaceFinding('## Historical example\nOld material.\n\n## Current findings\n' + row)).toBe(true); -}); - -test('new regression files select the existing SDK workflow owner', () => { - for (const file of ['test/sdk-ordered-schedule-ar.test.ts', 'test/fixtures/sdk-ordered-schedule-ar.md']) { - expect(selectTests([file], E2E_TOUCHFILES).selected).toEqual(['plan-ceo-section-loading']); - } -}); diff --git a/test/sdk-ordering-ae.test.ts b/test/sdk-ordering-ae.test.ts deleted file mode 100644 index 7ec20da6c..000000000 --- a/test/sdk-ordering-ae.test.ts +++ /dev/null @@ -1,83 +0,0 @@ -import {expect, test} from 'bun:test'; -import fixture from './fixtures/sdk-ordering-ae.json'; -import {hasStaleFillRaceFinding as found} from './helpers/ceo-section-loading-fixture'; -import {E2E_TOUCHFILES, selectTests} from './helpers/touchfiles'; - -const trace = fixture.f1.split('|')[4]!.trim(); -function withTrace(value: string): string { - const cells = fixture.f1.split('|'); - cells[4] = ` ${value} `; - return cells.join('|'); -} - -test('actual completed F1 report row supplies ordered stale-fill evidence without a race keyword', () => { - expect(found(fixture.f1)).toBe(true); - expect(fixture.provenance.historicalOutcome).toContain('timeout480032ms'); - expect(trace).not.toMatch(/\b(?:race|in-flight|concurrent|pending)\b/i); - expect(found(`F1 — P1: ${trace}`)).toBe(true); -}); - -test('ordering evidence requires miss, committed invalidation, stale refill and later stale readers', () => { - for (const value of [ - 'Reader fills the pre-commit snapshot; write commits and deletes; read misses; every later reader sees stale data.', - 'Read misses; reader then fills the pre-commit snapshot; write commits and deletes; every later reader sees stale data.', - 'Write commits and deletes; read misses; reader then fills the pre-commit snapshot; every later reader sees stale data.', - 'Read misses; reader then fills the pre-commit snapshot; every later reader sees stale data.', - 'Read misses, write commits; reader then fills the pre-commit snapshot; every later reader sees stale data.', - 'Read misses, write commits and deletes; every later reader sees stale data.', - 'Read misses, write commits and deletes; reader then fills the post-commit snapshot; every later reader sees fresh data.', - 'Read misses, write commits and deletes; the original reader returns its pre-commit snapshot to its own caller; every later reader sees fresh data.', - ]) expect(found(withTrace(value))).toBe(false); -}); - -test('explicit other cache, key or reader references cannot borrow the anonymous same-read trace', () => { - for (const value of [ - 'Read misses cache A, write commits and deletes cache B, reader then fills cache A with the pre-commit snapshot; every later reader sees stale data in cache A.', - 'Read misses key u1, write commits and deletes key u2, reader then fills key u1 with the pre-commit snapshot; every later reader sees stale data for key u1.', - 'Read R1 misses, write commits and deletes, reader R2 then fills the pre-commit snapshot; every later reader sees stale data.', - ]) expect(found(withTrace(value))).toBe(false); -}); - -test('hypothetical, negated and unestablished traces do not assert a current defect', () => { - for (const value of [ - `If ${trace[0]!.toLowerCase()}${trace.slice(1)}`, - `A hypothetical example: ${trace}`, - `An unproven hypothesis: ${trace}`, - `The following trace is impossible: ${trace}`, - `An unrelated illustration: ${trace}`, - `It is unclear whether this happens: ${trace}`, - `This trace did not occur: ${trace}`, - trace.replace('Read misses', 'Read may miss'), - trace.replace('write commits and deletes', 'write does not commit or delete'), - trace.replace('reader then fills', 'reader never fills'), - trace.replace('every later reader sees stale data', 'every later reader never sees stale data'), - 'Read misses, write commits and deletes, reader then fills the pre-commit snapshot; every later reader sees stale data?', - 'Read misses, write commits and deletes, reader then fills the pre-commit snapshot; every later reader sees stale data. This scenario is impossible.', - ]) expect(found(withTrace(value))).toBe(false); -}); - -test('copied source and independent rows or cells cannot supply missing ordered operations', () => { - for (const value of [`> ${fixture.f1}`, ` ${fixture.f1}`, `\t${fixture.f1}`, - `\`\`\`text\n${fixture.f1}\n\`\`\``, `~~~text\n${fixture.f1}\n~~~`]) expect(found(value)).toBe(false); - const first = withTrace('Read misses; write commits and deletes.'); - const last = withTrace('Reader then fills the pre-commit snapshot; every later reader sees stale data.').replace('| F1 |', '| F2 |'); - expect(found(first + '\n' + last)).toBe(false); - const cells = fixture.f1.split('|'); - cells[4] = ' Read misses; write commits and deletes. '; - cells[6] = ' Reader then fills the pre-commit snapshot; every later reader sees stale data. '; - expect(found(cells.join('|'))).toBe(false); - expect(found(first + '\n\n> ' + trace)).toBe(false); - expect(found(withTrace(`"${trace}" is a copied source example, not an observed defect.`))).toBe(false); -}); - -test('a real trace still rejects dismissal or acceptance of the later stale consequence', () => { - for (const suffix of [' No fix is required.', ' This stale-read behavior is accepted.', ' There is no stale-fill race.', - ' Later readers may return stale data and that is permitted.']) expect(found(withTrace(trace + suffix))).toBe(false); - expect(found(withTrace(trace + ' Original reader returns v1 to its own caller (allowed: it began before commit).'))).toBe(true); - expect(found(withTrace(trace + ' Later reader returns v1 to its own caller (allowed: it began after commit).'))).toBe(false); -}); - -test('new ordered-report evidence selects only the existing SDK section-loading case', () => { - for (const file of ['test/sdk-ordering-ae.test.ts', 'test/fixtures/sdk-ordering-ae.json']) - expect(selectTests([file], E2E_TOUCHFILES, []).selected).toEqual(['plan-ceo-section-loading']); -}); diff --git a/test/sdk-original-order-ai.test.ts b/test/sdk-original-order-ai.test.ts deleted file mode 100644 index db832cca3..000000000 --- a/test/sdk-original-order-ai.test.ts +++ /dev/null @@ -1,143 +0,0 @@ -import { describe, expect, test } from 'bun:test'; -import { hasStaleFillRaceFinding } from './helpers/ceo-section-loading-fixture'; -import captured from './fixtures/sdk-original-order-ai.json'; - -const compact = () => `### Findings registry\n\n${captured.finding}\n\n${captured.heading}\n\`\`\`\n${captured.trace}\n\`\`\``; -const rejects = (changes: Array<[string, string]>) => { - for (const [before, after] of changes) { - expect(compact()).toContain(before); - expect(hasStaleFillRaceFinding(compact().replace(before, after))).toBe(false); - } -}; - -describe('asserted original order beside an amended cache schedule', () => { - test('exact completed report and its owned finding/schedule show the original late-fill violation', () => { - expect(hasStaleFillRaceFinding(captured.report)).toBe(true); - expect(hasStaleFillRaceFinding(compact())).toBe(true); - }); - - test('amended behavior or the original caller allowance cannot replace the original stale-fill evidence', () => { - const original = 'Original sketch, order A: fill V1 at 6 after delete at 4 -> R2 reads V1 for <=30 s VIOLATION'; - rejects([ - [original, ''], [original, 'Not ' + original], [original, '> ' + original], - [original, '"' + original + '"'], [original, 'If ' + original], - [original, original.replace('VIOLATION', 'PERMITTED')], - [original, original.replace('R2 reads', 'R1 reads')], - [original, original.replace('fill V1', 'skip fill V1')], - [original, original.replace('after delete at 4', 'before delete at 4')], - ]); - }); - - test('reader, writer, cache key, versions and completion order must all refer to the same execution', () => { - rejects([ - ['inflight[k]', 'inflight[foreign]'], ['R1 (began before W)', 'R1 (began after W)'], - ['DB write commits V2', 'DB write commits V1'], ['DB returns V1', 'DB returns V2'], - ['resume: invalidate(E1), delete', 'resume: invalidate(E9), delete'], - ['settles -> W complete', 'settles -> W pending'], - ['resume: E1.stale -> skip fill', 'resume: E9.stale -> skip fill'], - ['7 | R2 begins:', '4.5 | R2 begins:'], ['R2 reads V1 for', 'R2 reads V2 for'], - ['fill V1 at 6 after delete at 4', 'fill V1 at 3 after delete at 4'], - ['cache[k]', 'cache[foreign]'], - ]); - }); - - test('the current finding owns the trace and must independently assert the invariant violation', () => { - rejects([ - ['schedule (F1,', 'schedule (F2,'], ['| F1 | CRITICAL |', '| F2 | CRITICAL |'], - ['| F1 | CRITICAL |', '| F1 | LOW |'], - ['Schedule in Section 4 shows', 'A hypothetical Schedule in Section 4 shows'], - ['filled after `cache.delete`', 'filled before `cache.delete`'], - ['every read begun after that write completes must observe the committed version', 'earlier values are accepted for later readers'], - ]); - expect(hasStaleFillRaceFinding(compact().replace(captured.finding, captured.finding + '\n' + captured.finding))).toBe(false); - }); - - test('source and hypothetical framing cannot supply the assertion', () => { - for (const prefix of ['An unproven hypothesis.', 'Historical example only.', 'The following is a hypothetical example.']) { - expect(hasStaleFillRaceFinding(prefix + '\n' + compact())).toBe(false); - expect(hasStaleFillRaceFinding(compact().replace(captured.heading, prefix + '\n' + captured.heading))).toBe(false); - } - expect(hasStaleFillRaceFinding(compact().split('\n').map(line => '> ' + line).join('\n'))).toBe(false); - expect(hasStaleFillRaceFinding('````text\n' + compact() + '\n````')).toBe(false); - expect(hasStaleFillRaceFinding(compact().replace('### Findings registry', '### Quoted source'))).toBe(false); - }); - - test('same-finding direct and quoted withdrawals remain authoritative inside or after the trace', () => { - for (const withdrawal of ['F1 is withdrawn.', 'F1 is rejected.', 'The original schedule is impossible.', 'There is no stale-fill race.', 'Rejected: "There is no stale-fill race."']) { - expect(hasStaleFillRaceFinding(compact() + '\n\n' + withdrawal)).toBe(false); - expect(hasStaleFillRaceFinding(compact().replace(captured.trace, captured.trace + '\n' + withdrawal))).toBe(false); - expect(hasStaleFillRaceFinding(compact().replace('Ordering tests, both orders + late joiner + sentinel variant', withdrawal))).toBe(false); - } - expect(hasStaleFillRaceFinding(compact().replace('Readers that began before the write may still see the old snapshot (permitted by contract)', 'Later readers may see old snapshots; this stale-fill behavior is accepted.'))).toBe(false); - }); - - test('unrelated sections and consistently renamed identities do not change valid evidence', () => { - expect(hasStaleFillRaceFinding('### Prior example\nHistorical example only.\n\n### Current review\n' + compact())).toBe(true); - expect(hasStaleFillRaceFinding(compact() + '\n\n### Other finding\nF2 is rejected.')).toBe(true); - const renamed = compact().replaceAll('R1', 'R7').replaceAll('R2', 'R8').replaceAll('R3', 'R9') - .replaceAll('V1', 'oldSnapshot').replaceAll('V2', 'newSnapshot').replaceAll('E1', 'pendingA').replaceAll('E2', 'pendingB') - .replaceAll('[k]', '[profileKey]').replace(/\bW\b/g, 'W2'); - expect(hasStaleFillRaceFinding(renamed)).toBe(true); - }); - - test('owning source headings and same-finding assessments survive intervening structure', () => { - for (const heading of ['## Hypothetical example', '## Quoted source', '## Historical example only']) { - expect(hasStaleFillRaceFinding(heading + '\n' + compact())).toBe(false); - } - expect(hasStaleFillRaceFinding(compact().replace(captured.heading, - 'F1 is rejected.\n\nUnrelated diagram:\n```\nA -> B\n```\n\n' + captured.heading))).toBe(false); - expect(hasStaleFillRaceFinding(compact() + '\n\n### Assessment of F1\nF1 is rejected.')).toBe(false); - }); -}); - -const retry = () => `## Findings Registry\n\n${captured.retry.finding}\n\n${captured.retry.heading}\n\`\`\`\n${captured.retry.trace}\n\`\`\``; -describe('version-labelled original prose with its owned schedule', () => { - test('the exact retry and compact evidence require the original sequence, not amended prevention', () => { - expect(hasStaleFillRaceFinding(captured.retry.report)).toBe(true); - expect(hasStaleFillRaceFinding(retry())).toBe(true); - }); - - test('each version and shared key must agree, with write completion before the later reader', () => { - for (const [before, after] of [ - ['DB returns v1', 'DB returns v2'], ['write commits v2 and', 'write commits v1 and'], - ['read then fills v1;', 'read then fills v2;'], ['every later read gets v1', 'every later read gets v2'], - ['write commits v2 and', 'write commits v3 and'], ['writeGen[key]', 'writeGen[foreign]'], - ['cache[key]', 'cache[foreign]'], ['R2 (read, began after W)', 'R2 (read, began before W)'], - ['delete (no-op), return', 'delete (no-op), pending'], ['DB SELECT -> v1', 'DB SELECT -> v2'], - ['DB UPDATE commits v2', 'DB UPDATE commits v3'], ['promise resolves, set(v1)', 'promise resolves, set(v2)'], - ['get -> v1 VIOLATION', 'get -> v2 VIOLATION'], ['6 sketch', '3 sketch'], - ['3 DB UPDATE commits v2', '3 DB UPDATE commits v2'], - ]) { - expect(retry()).toContain(before); - expect(hasStaleFillRaceFinding(retry().replace(before, after))).toBe(false); - } - expect(hasStaleFillRaceFinding(retry().replace(captured.retry.trace, captured.retry.trace.split('\n').filter(line => !/\b[456] sketch\b/.test(line)).join('\n')))).toBe(false); - }); - - test('conditional, quoted, obsolete or withdrawn evidence cannot become a current finding', () => { - for (const prefix of ['An unproven hypothesis.', 'Historical example only.', 'The following is a hypothetical example.']) { - expect(hasStaleFillRaceFinding(prefix + '\n' + retry())).toBe(false); - expect(hasStaleFillRaceFinding(retry().replace('Late fill after write.', prefix + ' Late fill after write.'))).toBe(false); - } - for (const heading of ['## Hypothetical example', '## Quoted source', '## Historical example only']) { - expect(hasStaleFillRaceFinding(heading + '\n' + retry().replace('## Findings Registry', '### Findings Registry'))).toBe(false); - } - for (const withdrawal of ['F1 is rejected.', 'S1 is withdrawn.', 'The original schedule is impossible.', 'There is no stale-fill race.', 'Rejected: "There is no stale-fill race."']) { - expect(hasStaleFillRaceFinding(retry() + '\n\n' + withdrawal)).toBe(false); - expect(hasStaleFillRaceFinding(retry() + '\n\n### Assessment of F1\n' + withdrawal)).toBe(false); - expect(hasStaleFillRaceFinding(retry().replace(' S2 join stale flight', withdrawal + '\n S2 join stale flight'))).toBe(false); - } - for (const withdrawal of ['S1 is withdrawn.', 'F1 is rejected.']) { - expect(hasStaleFillRaceFinding(retry().replace(captured.retry.trace, captured.retry.trace + '\n' + withdrawal))).toBe(false); - } - expect(hasStaleFillRaceFinding(retry().split('\n').map(line => '> ' + line).join('\n'))).toBe(false); - expect(hasStaleFillRaceFinding('````\n' + retry() + '\n````')).toBe(false); - expect(hasStaleFillRaceFinding(retry().replace('Late fill after write.', 'If a late fill happens after write.'))).toBe(false); - }); - - test('consistent versions and independent later findings remain valid', () => { - expect(hasStaleFillRaceFinding(retry().replaceAll('v1', 'v7').replaceAll('v2', 'v8').replaceAll('[key]', '[profileKey]').replaceAll('key#1', 'profileKey#1'))).toBe(true); - expect(hasStaleFillRaceFinding('## Prior example\nHistorical only.\n\n## Current review\n' + retry().replace('## Findings Registry', '### Findings Registry'))).toBe(true); - expect(hasStaleFillRaceFinding(retry() + '\n\n### Other finding\nF9 is rejected.')).toBe(true); - }); -}); diff --git a/test/sdk-reported-coordination-ar.test.ts b/test/sdk-reported-coordination-ar.test.ts deleted file mode 100644 index 453c064c4..000000000 --- a/test/sdk-reported-coordination-ar.test.ts +++ /dev/null @@ -1,69 +0,0 @@ -import { expect, test } from 'bun:test'; -import fs from 'node:fs'; -import { hasStaleFillRaceFinding } from './helpers/ceo-section-loading-fixture'; -import { E2E_TOUCHFILES, selectTests } from './helpers/touchfiles'; - -const report = fs.readFileSync(new URL('./fixtures/sdk-reported-coordination-ar.md', import.meta.url), 'utf8'); -const paragraph = report.split('\n\n').find(text => text.startsWith('## Proposed wrapper integration'))!.split('\n').slice(1).join('\n'); -const matches = (text = paragraph) => hasStaleFillRaceFinding(text); - -test('the actual retry independently reports the original coordination violation', () => { - expect(paragraph).toContain('review found that this violates the read-after-write rule above (F1)'); - expect(paragraph).not.toMatch(/stale|in-flight|race|pending/); - expect(matches()).toBe(true); - expect(matches(report)).toBe(true); - expect(matches(paragraph.replace('proposed no coordination', 'had no coordination'))).toBe(true); - expect(matches(paragraph.replace('proposed no coordination', 'has no coordination'))).toBe(true); - expect(matches(paragraph.replace('sketch', 'wrapper'))).toBe(true); - expect(matches(paragraph.replace('rule above', 'contract'))).toBe(true); - expect(matches(paragraph.replace(/ and omits[\s\S]*/, '.'))).toBe(true); -}); - -test('missing or hypothetical premise and conclusion cannot become findings', () => { - for (const [from, to] of [ - ['proposed no coordination', 'proposed coordination'], - ['proposed no coordination', 'may propose no coordination'], - ['review found that this violates', 'review may find that this violates'], - ['review found that this violates', 'review found that this does not violate'], - ['review found that this violates', 'review hypothesized that this violates'], - ['review found that this violates', 'review found that another wrapper violates'], - ['read-after-write rule above', 'formatting rule'], - ['(F1)', '(unknown)'], - ['; the\nreview found', '. Another unrelated finding. The\nreview found'], - ['; the\nreview found', '\n\nThe\nreview found'], - ['; the\nreview found', ' | The\nreview found'], - ]) { - expect(paragraph).toContain(from); - expect(matches(paragraph.replace(from, to))).toBe(false); - } -}); - -test('source and quoted evidence cannot assert the current violation', () => { - for (const text of [ - 'Source:\n\n' + paragraph, - 'Hypothetical scenario. ' + paragraph, - 'Earlier review:\n\n' + paragraph, - '## Historical example\n' + paragraph, - '> ' + paragraph.replaceAll('\n', '\n> '), - '```text\n' + paragraph + '\n```', - '~~~text\n' + paragraph + '\n~~~', - paragraph.replace('original sketch proposed no coordination between a cache fill and a write', '`original sketch proposed no coordination between a cache fill and a write`'), - paragraph.replace('review found that this violates the read-after-write rule above (F1)', '"review found that this violates the read-after-write rule above (F1)"'), - ]) expect(matches(text)).toBe(false); -}); - -test('the referenced finding owns its later assessment', () => { - for (const tail of ['F1 is withdrawn.', 'F1 is "withdrawn".', 'F1 is rejected.', 'This finding is dismissed.', 'No coordination is required.', '| ID | Assessment |\n| F1 | Withdrawn: no coordination is required. |', '| F1 | Withdrawn |', '| F1 | "rejected" |']) { - expect(matches(paragraph + '\n\n' + tail)).toBe(false); - } - expect(matches(paragraph + '\n\nF2 is withdrawn.')).toBe(true); - expect(matches(paragraph + '\n\n| F2 | Withdrawn |')).toBe(true); - expect(matches(paragraph + '\n\n## Historical assessment\n| F1 | Withdrawn |')).toBe(true); - expect(matches('## Earlier material\nSource:\nOld source.\n\n## Current findings\n' + paragraph)).toBe(true); -}); - -test('the new regression selects its existing SDK workflow owner', () => { - for (const file of ['test/sdk-reported-coordination-ar.test.ts', 'test/fixtures/sdk-reported-coordination-ar.md']) { - expect(selectTests([file], E2E_TOUCHFILES).selected).toEqual(['plan-ceo-section-loading']); - } -}); diff --git a/test/sdk-schedule-continuation-ah.test.ts b/test/sdk-schedule-continuation-ah.test.ts deleted file mode 100644 index 6a3e3a8c5..000000000 --- a/test/sdk-schedule-continuation-ah.test.ts +++ /dev/null @@ -1,121 +0,0 @@ -import { expect, test } from 'bun:test'; -import fixture from './fixtures/sdk-schedule-continuation-ah.json'; -import { hasStaleFillRaceFinding } from './helpers/ceo-section-loading-fixture'; -import { E2E_TOUCHFILES } from './helpers/touchfiles-data'; -import { selectTests } from './helpers/touchfiles'; - -const frame = fixture.compact; -function replace(from: string, to: string, input = frame): string { - expect(input.includes(from)).toBe(true); - return input.replace(from, to); -} -const originalRows = ' S2* | await read ... | write commits, delete(noop) | | - |\n' - + ' | resolves V0 → set V0 | | hit → V0 | V0 (30 s) | VIOLATION\n'; - -test('retains both exact public report forms as affirmative original-race findings', () => { - expect(hasStaleFillRaceFinding(fixture.report)).toBe(true); - expect(hasStaleFillRaceFinding(frame)).toBe(true); - expect(fixture.report.includes(frame.trim())).toBe(true); -}); - -test('binds consistently renamed actors, shared key, versions and finding/schedule IDs', () => { - const renamed = frame.replace(/\bR1\b/g, 'R7').replace(/\bR2\b/g, 'R8').replace(/\bW\b/g, 'W9') - .replace(/\bV0\b/g, 'oldValue').replace(/\bV1\b/g, 'freshValue') - .replace(/\bkey\b/g, 'profile_key').replace(/\bF1\b/g, 'F9').replace(/\bS2\b/g, 'S9'); - expect(hasStaleFillRaceFinding(renamed)).toBe(true); - expect(hasStaleFillRaceFinding(frame.replace(/→/g, '->'))).toBe(true); - const unrelated = '## Historical example\nAn unrelated old example.\n\n## Current findings\n\n'; - expect(hasStaleFillRaceFinding(unrelated + frame)).toBe(true); -}); - -test('amendments, permitted earlier readers and missing continuation do not supply the original race', () => { - for (const changed of [ - replace(originalRows, ''), - replace('S2* | await read', 'S2 A1 | await read'), - replace(originalRows, ' S2* | begins before W, joins | delete + forget | — | — | OK: R1 began before W completed (permitted clause)\n'), - replace('resolves V0 → set V0', 'resolves V0, slot gone→drop'), - replace('hit → V0', 'miss→read V1→set'), - replace('hit → V0', ''), - replace('VIOLATION\n S2 A1', 'OK (permitted earlier return)\n S2 A1'), - replace('VIOLATION\n S2 A1', 'VIOLATION\n | already guarded | | | | OK\n S2 A1'), - ]) expect(hasStaleFillRaceFinding(changed)).toBe(false); -}); - -test('requires the original schedule citation, legend and explicit post-completion boundary', () => { - for (const changed of [ - replace('Schedule S2 makes', 'Schedule S9 makes'), - replace('`*` = original sketch.', '`*` = amended sketch.'), - replace('`*` = original sketch.', ''), - replace('`*` = original sketch.', 'Hypothetically, `*` = original sketch.'), - replace('CRITICAL GAP | 1, 2, 4, 5, 6', 'CRITICAL GAP | 1, 2, 5, 6'), - replace('Violates retained invariant.', 'No defect in the retained invariant.'), - replace('R2 (begins after W)', 'R2 (begins before W)'), - replace('after `writeProfile` resolves', 'before `writeProfile` resolves'), - replace('after `writeProfile` resolves', 'after `writeProfile` begins'), - replace('after `writeProfile` resolves', 'after `readProfile` resolves'), - ]) expect(hasStaleFillRaceFinding(changed)).toBe(false); -}); - -test('rejects actor, key, value, invalidation and ordering mismatches', () => { - for (const changed of [ - replace('R2 (begins after W)', 'R1 (begins after W)'), - replace('R2 (begins after W)', 'R2 (begins after W9)'), - replace('`inflight[key]`', '`inflight[other_key]`'), - replace('| cache[key] | Result', '| cache[other_key] | Result'), - replace('W (commits V1)', 'W (commits V0)'), - replace('resolves V0 → set V0', 'resolves V1 → set V0'), - replace('resolves V0 → set V0', 'resolves V0 → set V1'), - replace('hit → V0', 'hit → V1'), - replace('write commits, delete(noop)', 'write begins, delete(noop)'), - replace('write commits, delete(noop)', 'write commits'), - replace('await read ...', 'await write ...'), - replace(originalRows, originalRows.split('\n').slice(0, 2).reverse().join('\n') + '\n'), - replace('V0 (30 s) | VIOLATION', 'V1 (30 s) | VIOLATION'), - replace('see V0 for 30 s;', 'see V1 for 30 s;'), - ]) expect(hasStaleFillRaceFinding(changed)).toBe(false); -}); - -test('quotes, source introductions and withdrawn findings remain negative', () => { - for (const prefix of ['An unproven hypothesis.', 'Historical example only.', 'The following is a hypothetical example.']) { - expect(hasStaleFillRaceFinding(prefix + '\n\n' + frame)).toBe(false); - expect(hasStaleFillRaceFinding(replace('### Async Ordering Record', prefix + '\n\n### Async Ordering Record'))).toBe(false); - expect(hasStaleFillRaceFinding(replace('### Findings Registry\n', '### Findings Registry\n\n' + prefix))).toBe(false); - } - expect(hasStaleFillRaceFinding(frame.split('\n').map(line => '> ' + line).join('\n'))).toBe(false); - expect(hasStaleFillRaceFinding('````text\n' + frame + '\n````')).toBe(false); - expect(hasStaleFillRaceFinding(replace('```\n Sched', '```javascript\n Sched'))).toBe(false); - for (const dismissal of [ - 'The original trace is impossible.', 'This schedule is not a bug.', - 'The original race is permitted.', 'The stale fill is accepted.', - 'No coordination is required.', - ]) { - expect(hasStaleFillRaceFinding(frame + '\n' + dismissal)).toBe(false); - expect(hasStaleFillRaceFinding(replace('Violates retained invariant.', 'Violates retained invariant. ' + dismissal))).toBe(false); - } -}); - - -test('completed prior decision section is independent; spoofed or withdrawn framing is not', () => { - const close = '### Decision Registry (all auto-resolved to recommended option)\n\n| D1 | A | B |\n\nLake Score: 7/7 recommendations chose the complete option.\n\n'; - expect(hasStaleFillRaceFinding(close + frame)).toBe(true); - expect(hasStaleFillRaceFinding(close.replace('### Decision Registry (all auto-resolved to recommended option)', '### Historical example') + frame)).toBe(false); - expect(hasStaleFillRaceFinding(close.replace('Lake Score: 7/7 recommendations chose the complete option.', 'An unproven hypothesis.') + frame)).toBe(false); - const row = frame.split('\n').find(line => line.startsWith('| F1 |'))!; - expect(hasStaleFillRaceFinding(replace(row, row + '\n' + row))).toBe(false); -}); - -test('only the existing SDK paid owner selects the added fixture and controls', () => { - for (const file of ['test/sdk-schedule-continuation-ah.test.ts', 'test/fixtures/sdk-schedule-continuation-ah.json']) { - expect(Object.entries(E2E_TOUCHFILES).filter(([, files]) => files.includes(file)).map(([name]) => name)).toEqual(['plan-ceo-section-loading']); - expect(selectTests([file], E2E_TOUCHFILES, []).selected).toEqual(['plan-ceo-section-loading']); - } -}); - - -test('same finding or schedule tail withdrawals remain authoritative', () => { - for (const tail of ['S2 is impossible.', 'F1 is rejected. The original trace is impossible.', 'F1 is rejected.', 'S2 is withdrawn.']) { - expect(hasStaleFillRaceFinding(frame + '\n' + tail)).toBe(false); - } - expect(hasStaleFillRaceFinding(frame + '\nF2 is rejected. The original trace is impossible.')).toBe(true); - expect(hasStaleFillRaceFinding(frame + '\nS3 is impossible.')).toBe(true); -}); diff --git a/test/sdk-stale-table-ad-v3.test.ts b/test/sdk-stale-table-ad-v3.test.ts deleted file mode 100644 index 0e694acea..000000000 --- a/test/sdk-stale-table-ad-v3.test.ts +++ /dev/null @@ -1,73 +0,0 @@ -import {expect,test} from 'bun:test'; -import fixture from './fixtures/sdk-stale-table-ad-v3.json'; -import {hasStaleFillRaceFinding as found} from './helpers/ceo-section-loading-fixture'; -const allowance='Original reader still returns v1 to its own caller (allowed: it began before commit)'; -test('actual table finding distinguishes forbidden later stale reads from the permitted original caller',()=>{ - expect(found(fixture.report)).toBe(true); - expect(found(fixture.table)).toBe(true); - expect(fixture.provenance.noRetroactivePass).toBe(true); -}); -test('the already-started original read may use a version label without changing ownership',()=>{ - for(const token of ['v17','VERSION_A','snapshot-A'])expect(found(fixture.table.replaceAll('v1',token))).toBe(true); -}); -test('allowance cannot migrate to later readers, a post-commit start, or a cache fill',()=>{ - for(const changed of [ - 'Later readers return v1 (allowed: they began after commit)', - 'Original reader still returns v1 to its own caller (allowed: it began after commit)', - 'Original reader still returns v1 to its own caller (allowed: it never began before commit)', - 'Original reader fills the cache with v1 (allowed: it began before commit)', - 'Original reader still returns v1 to later readers (allowed: it began before commit)', - ])expect(found(fixture.table.replace(allowance,changed))).toBe(false); -}); -test('a permitted original caller cannot hide acceptance of later stale reads or no required fix',()=>{ - for(const suffix of [' This stale-read behavior is accepted.',' No fix is required.',' Later readers may return stale data; this is the accepted consistency model.']) - expect(found(fixture.table.replace('None against the invariant.','None against the invariant.'+suffix))).toBe(false); -}); -test('copied table source and absent late-fill evidence cannot provide coverage',()=>{ - expect(found('```text\n'+fixture.table+'\n```')).toBe(false); - expect(found(fixture.table.split('\n').map(x=>'> '+x).join('\n'))).toBe(false); - expect(found(fixture.table.split('\n').map(x=>' '+x).join('\n'))).toBe(false); - const rows=fixture.table.split('\n'),cells=rows[2]!.split('|'); - cells[4]=' There is no stale-fill race; later reads observe the committed value. '; - rows[2]=cells.join('|');expect(found(rows.join('\n'))).toBe(false); -}); - - -test('original-caller exception requires asserted chronology for that reader',()=>{ - for(const changed of [ - 'Original reader still returns v1 to its own caller (allowed: it may have begun before commit)', - 'Original reader still returns v1 to its own caller (allowed: it did not begin before commit)', - 'Original reader still returns v1 to its own caller (allowed: it began before commit only if the write failed)', - 'Original reader still returns v1 to its own caller (allowed: another reader began before commit)', - 'Original reader still returns v1 to its own caller (allowed: the write began before commit)', - 'If the original reader still returns v1 to its own caller, that is allowed: it began before commit', - ])expect(found(fixture.table.replace(allowance,changed))).toBe(false); -}); - -test('an original-return allowance cannot erase another allowed stale consequence',()=>{ - for(const changed of [ - allowance+' and stores that v1 in the cache for later readers', - allowance+'; later readers may reuse this old value and that is allowed', - allowance+'. New readers may reuse this old value and that is permitted', - allowance+'. The stale cache refill is acceptable', - ])expect(found(fixture.table.replace(allowance,changed))).toBe(false); -}); - -test('table rows cannot borrow an ordering defect from another issue or from quoted source',()=>{ - const rows=fixture.table.split('\n'),cells=rows[2]!.split('|'); - const originalFailure=cells[4]!; - cells[4]=' The original reader receives its pre-commit snapshot; later reads observe the committed version. '; - const missing=rows.slice(0,2).concat(cells.join('|')).join('\n'); - expect(found(missing)).toBe(false); - const other=cells.slice();other[1]=' D2 ';other[4]=originalFailure; - other[5]=' This stale-read behavior is accepted; no fix is required. '; - expect(found(missing+'\n'+other.join('|'))).toBe(false); - expect(found('> '+originalFailure+'\n\n'+missing)).toBe(false); - expect(found('```text\n'+originalFailure+'\n```\n\n'+missing)).toBe(false); -}); - -import {E2E_TOUCHFILES,selectTests} from './helpers/touchfiles'; -test('the new table evidence selects only the existing SDK section-loading case',()=>{ - for(const file of ['test/sdk-stale-table-ad-v3.test.ts','test/fixtures/sdk-stale-table-ad-v3.json']) - expect(selectTests([file],E2E_TOUCHFILES,[]).selected).toEqual(['plan-ceo-section-loading']); -});