test(ceo-mode-routing): submit a mode review that scrolled past the viewport

Run 36606688266 bundled routing, learnings and the mode choice into one
native call. Its review panel was taller than the terminal, so the tab
bar scrolled off, ceoModeSubmissionInput returned null for 240 s and HOLD
SCOPE was never submitted ('no posture match'). With no bar on screen the
viewport must still end at the focused Submit prompt, and the accumulated
screen text supplies the one complete panel, authenticated exactly as
before. Replay controls reject another mode, an unoffered answer, an
altered question, a quoted panel, trailing output, a moved cursor and an
answered or changed call.
This commit is contained in:
garrytan committed 2026-09-29 19:16:01 +00:00
1 parent 69cb4c8e9d
commit 49761c97d6
4 files changed
+148 -11

No files matched your search

+53
View File
@@ -10,6 +10,7 @@ import captured_ceo_hold_commitment_ar from './fixtures/ceo-hold-commitment-ar.j
import captured_ceo_hold_posture_ag from './fixtures/ceo-hold-posture-ag.json';
import retainedPreservationCaptures_ceo_hold_posture_ag from './fixtures/ceo-hold-preservation-f359.json';
import captured_ceo_mode_colon_at from './fixtures/ceo-mode-colon-at.json';
import scrolledReview from './fixtures/ceo-mode-scrolled-review-36606688266.json';
import fs_ceo_mode_full_ad from 'node:fs';
import os_ceo_mode_full_ad from 'node:os';
import path_ceo_mode_full_ad from 'node:path';
@@ -1840,3 +1841,55 @@ test('AD v2 prerequisite requires the active native packet identity',()=>{
for(const delta of [{answered:true},{failed:true},{sessionId:''},{toolUseId:''}]){const call={...pending(),...delta};const x=frame(call,2);expect(planCountPrerequisitePick(x.routing,x.active)).toBeNull();}
});
});
describe('mode submission when the review panel scrolls past the viewport', () => {
// Run 36606688266 bundled routing, learnings and the mode choice into one
// native call. Its review panel was taller than the terminal, so the tab bar
// scrolled away and the harness never submitted HOLD SCOPE.
const scrolledTranscript = scrolledReview.transcript as unknown as PlanCountTranscript;
const scrolledCall = scrolledTranscript.calls[0] as NativePlanQuestionCall;
const scrolledSubmit = (screen: string, screenText: string, mode: 'HOLD SCOPE' | 'SCOPE EXPANSION' = 'HOLD SCOPE',
selected: NativePlanQuestionCall = scrolledCall, native: PlanCountTranscript = scrolledTranscript) =>
ceoModeSubmissionInput(screen, selected, mode, native, new Set(), screenText);
test('the captured viewport has no tab bar and ends at the focused Submit prompt', () => {
expect(scrolledReview.screen).not.toMatch(/←[^\r\n]+✔\s*Submit\s*→/);
expect(scrolledReview.screen.trimEnd()).toMatch(/❯ 1\. Submit answers\s+2\. Cancel$/);
expect(scrolledCall.questions.map(q => q.header)).toEqual(['Routing', 'Learnings', 'Review mode']);
});
test('the complete scrolled review submits the selected mode once', () => {
expect(scrolledSubmit(scrolledReview.screen, scrolledReview.screenText)).toBe('\r');
const seen = new Set<string>();
expect(ceoModeSubmissionInput(scrolledReview.screen, scrolledCall, 'HOLD SCOPE', scrolledTranscript, seen, scrolledReview.screenText)).toBe('\r');
expect(ceoModeSubmissionInput(scrolledReview.screen, scrolledCall, 'HOLD SCOPE', scrolledTranscript, seen, scrolledReview.screenText)).toBeNull();
});
test('without the accumulated screen text a barless viewport cannot submit', () => {
expect(scrolledSubmit(scrolledReview.screen, '')).toBeNull();
});
test('a review showing another mode is not an acknowledgement of the target mode', () => {
expect(scrolledSubmit(scrolledReview.screen, scrolledReview.screenText, 'SCOPE EXPANSION')).toBeNull();
});
for (const [name, change] of [
['an answer no option offers', (text: string) => text.replace(/→ Enable cross-project \(recommended\)(?![\s\S]*→ Enable cross-project)/, '→ Upload learnings')],
['an altered question', (text: string) => text.replace(/D2 — Let gstack(?![\s\S]*D2 — Let gstack)/, 'D2 — Never let gstack')],
['a quoted review', (text: string) => text.replace(/Review your answers(?![\s\S]*Review your answers)/, 'Quoted example:\nReview your answers')],
['output after the prompt', (text: string) => `${text}\nMore text`],
] as const) test(`the scrolled route rejects ${name}`, () => {
expect(scrolledSubmit(scrolledReview.screen, change(scrolledReview.screenText))).toBeNull();
});
test('the viewport must still end at the focused Submit prompt', () => {
expect(scrolledSubmit(scrolledReview.screen.replace('❯ 1. Submit answers', ' 1. Submit answers\n❯ 2. Cancel'), scrolledReview.screenText)).toBeNull();
});
test('an answered or changed native call cannot be submitted again', () => {
expect(scrolledSubmit(scrolledReview.screen, scrolledReview.screenText, 'HOLD SCOPE', { ...scrolledCall, answered: true })).toBeNull();
const other = structuredClone(scrolledCall);
other.questions[1]!.question += ' (changed)';
expect(scrolledSubmit(scrolledReview.screen, scrolledReview.screenText, 'HOLD SCOPE', other)).toBeNull();
});
});
+72
View File
@@ -0,0 +1,72 @@
{
"source": "run 36606688266 plan-ceo-mode-routing HOLD SCOPE: final viewport, accumulated screen text from the last tab frame, and the pending native call",
"screen": " \u2502 ELI10: gstack works best when your project's CLAUDE.md includes skill routing rules, so requests like \"review this\n \u2502 diff\" route to the right skill automatically. This is a plain text section appended to CLAUDE.md.\n \u2502 Stakes if we pick wrong: without it you invoke skills by name manually; with it, plain requests auto-route. Either\n \u2502 is reversible.\n \u2502 Recommendation: A because auto-routing removes a step from every future session and costs one commit.\n \u2502 Note: options differ in kind, not coverage \u2014 no completeness score.\n \u2502 Net: convenience now vs. one extra committed section in CLAUDE.md. Plan mode blocks file edits, so if you pick A\n \u2502 the append + commit happens after this review exits plan mode.\n \u2192 Add routing rules (recommended)\n \u2502 \u25cf D2 \u2014 Let gstack search learnings from your other projects on this machine?\n \u2502 Project/branch/task: gstack-plan-count-FwyQuk on main; one-time gstack setup prompt.\n \u2502 ELI10: gstack saves small lessons per project (\"this test runner needs flag X\"). Cross-project mode also searches\n \u2502 lessons from your other local projects when reviewing this one. Everything stays on this machine.\n \u2502 Stakes if we pick wrong: too narrow and you miss patterns you already learned elsewhere; too broad and a client\n \u2502 codebase could surface a lesson from another client's repo in a review.\n \u2502 Recommendation: A because this is a solo-style environment and the data never leaves the machine.\n \u2502 Note: options differ in kind, not coverage \u2014 no completeness score.\n \u2502 Net: more recall vs. strict per-project isolation.\n \u2192 Enable cross-project (recommended)\n \u2502 \u25cf D3 \u2014 R2: Which review mode for the saved-views plan?\n \u2502 Project/branch/task: gstack-plan-count-FwyQuk on main; reviewing PLAN.md \"Add saved project views\".\n \u2502 ELI10: The mode sets my posture for the rest of the review. Expansion pushes for the biggest version, Hold Scope\n \u2502 stress-tests exactly what you wrote, Reduction strips to the smallest shippable core, and Selective holds your\n \u2502 scope while offering a few add-ons one at a time for you to accept or decline.\n \u2502 Stakes if we pick wrong: too ambitious and a small feature balloons; too strict and we ship personal-only views\n \u2502 when the goal (\"team members repeatedly recreate filters\") may really be a shared-view problem, forcing a second\n \u2502 migration later.\n \u2502 Recommendation: SELECTIVE EXPANSION because the plan is an added capability of ~8\u201310 files, but its member-only\n \u2502 scoping is the one fact that could be wrong: every incumbent ships shared views too, and the table shape decides\n \u2502 whether adding them later is a column or a rewrite. Selective lets you rule on that once without committing to a\n \u2502 bigger build.\n \u2502 Note: options differ in kind, not coverage \u2014 no completeness score.\n \u2502 Net: how much of the review is spent challenging scope vs. hardening the scope you already chose.\n \u2192 HOLD SCOPE\n\nReady to submit your answers?\n\n\u276f 1. Submit answers\n 2. Cancel\n",
"screenText": "omething.\n\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n 4. Chat about this\n\nEnter to select \u00b7 Tab/Arrow keys to navigate \u00b7 Esc to cancel\n\n\n\n \u2612 Learnings \u2610 Review mode \n3R2:Which review mode for the saved-views plan?\nreviewing PLAN.md \"Add saved project views\".\nThe mode sts y posture for the rest of e review. Expansion pushes forthe biggest version, Hld Scop \nstress-testsexactly what youwt, Reduction strpso the smallst shippable cre, and Selective holds your scope \nwhil ofering a few add-onsone at time for you o accept or dcline.\nStakes if we pick wrong:too ambitious and asmall featureballoons; too strict and we ship personal-only views when \nthe goal (\"teammembers repeatedly recreate filters\") ay really be shared-viw problm, forcing a second migration \nlar.\nRcomendation: SELECTIVE EXPANSION becaue the plan is an added capability of ~8\u201310 files, but its member-only \n\u2502scoping is the one fact that could be wrong: every incumbent ships shared views too, and the table shape decides \n\u2502whether adding them later is a column or a rewrite. Selective lets you rule on that once without committing to a \n\u2502bigger build.\n\u2502Note: options differ in kind, not coverage \u2014 no completeness score.\n\u2502Net: how much of the review is spent challenging scope vs. hardening the scope you already chose.\n\n\u276f1.SELECTIVE EXPANSION (recommended)\n\u2705 Keep your fou approach bullets as the bselineand hardens hemwith fullrigor\ufffd\u2705 Offers each expansion \n (shared views, default view, cleanup) as a separate add/defer/skip call\ufffd\u274c A few more decision questions than Hold \n Scope before the deep review starts\n2.HOLDSCOPE\n\u2705 Maximum rigor on exactly what is written: error paths, edge cases, tests, observability\ufffd\u2705 Fastest path to an \n implementation-ready plan wiho scopquetions\ufffd\u274c Shared views and table-shape futureproofing get flagged, not \noffered; possible second migration later\n\n3.SCOPEEXPANSION\n\n\u2705Designstheplatonicsaved-viewsfeature:personal+shared,defaults,sharelinks,cleanup\ufffd\u2705Bestlong-term\n\narchitectureupfront;nofollow-upmigrations\ufffd\u274cTurnsa~10-filefeatureintoamulti-surfacebuildbeforethe\n\ntwo-weekpilotprovesreuse\n\n4.SCOPEREDUCTION\n\n\u2705Findsthesmallestcorethatteststhepilothypothesis(maybecreate/list/applyonly)\ufffd\u2705Lowestriskand\n\nfastesttothetwo-weekreusemeasurement\ufffd\u274cUpdate/deleteandpickerpolishgetdeferred;pilotmaymeasurea\n\nclunkyversionofthefeature\n\n5.Typesomething.\n\n\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n\n6.Chataboutthis\n\n\n\nEntertoselect\u00b7Tab/Arrowkeystonavigate\u00b7Esctocancel\n\n\n\nReview your answers\n \u2502 \u25cf D1 \u2014 Add gstack skill routing rules to this project'sCLAUDE.md?\n \u2502 Project/branch/task: gstack-plan-count-FwyQuk on main; one-time gstack setup prompt.\n \u2502 ELI10: gstack works best when your project's CLAUDE.md includes skill routing rules, so requests like \"review this\n \u2502 diff\" route to the right skill automatically. This is a plain text section appended to CLAUDE.md.\n \u2502 Stakes if we pick wrong: without it you invoke skills by name manually;withit,plainrequestsauto-route.Either\n \u2502 is reversible.\n \u2502 Recommendation: A because auto-routing removes a step from every future session and costs one commit.\n \u2502 Note:optionsdifferinkind,notcoverage\u2014nocompletenessscore.\n \u2502 Net: convenience now vs. one extra committed section in CLAUDE.md. Plan mode blocks file edits, so if you pickA\n \u2502 the appenLine truncated
"transcript": {
"status": "ready",
"calls": [
{
"sessionId": "4ea226bf-5a31-4342-8141-c76094320f94",
"toolUseId": "toolu_01Ee4tS4SoMXqLyS14ayssEy",
"questions": [
{
"question": "D1 \u2014 Add gstack skill routing rules to this project's CLAUDE.md?\nProject/branch/task: gstack-plan-count-FwyQuk on main; one-time gstack setup prompt.\nELI10: gstack works best when your project's CLAUDE.md includes skill routing rules, so requests like \"review this diff\" route to the right skill automatically. This is a plain text section appended to CLAUDE.md.\nStakes if we pick wrong: without it you invoke skills by name manually; with it, plain requests auto-route. Either is reversible.\nRecommendation: A because auto-routing removes a step from every future session and costs one commit.\nNote: options differ in kind, not coverage \u2014 no completeness score.\nNet: convenience now vs. one extra committed section in CLAUDE.md. Plan mode blocks file edits, so if you pick A the append + commit happens after this review exits plan mode.",
"header": "Routing",
"multiSelect": false,
"options": [
{
"label": "Add routing rules (recommended)",
"description": "\u2705 Plain requests (\"ship this\", \"find the bug\") route to the right skill without naming it\n\u2705 One short section, committed once; teammates get the same routing\n\u274c Adds a gstack-specific block to CLAUDE.md; applied after plan mode exits, not now"
},
{
"label": "No thanks, manual invocation",
"description": "\u2705 CLAUDE.md stays exactly as it is; no tool-specific text in the repo\n\u2705 You keep full control over when a skill runs\n\u274c Every skill needs its /name typed; re-enable later via gstack-config routing_declined false"
}
]
},
{
"question": "D2 \u2014 Let gstack search learnings from your other projects on this machine?\nProject/branch/task: gstack-plan-count-FwyQuk on main; one-time gstack setup prompt.\nELI10: gstack saves small lessons per project (\"this test runner needs flag X\"). Cross-project mode also searches lessons from your other local projects when reviewing this one. Everything stays on this machine.\nStakes if we pick wrong: too narrow and you miss patterns you already learned elsewhere; too broad and a client codebase could surface a lesson from another client's repo in a review.\nRecommendation: A because this is a solo-style environment and the data never leaves the machine.\nNote: options differ in kind, not coverage \u2014 no completeness score.\nNet: more recall vs. strict per-project isolation.",
"header": "Learnings",
"multiSelect": false,
"options": [
{
"label": "Enable cross-project (recommended)",
"description": "\u2705 Lessons from other local projects surface here when they match a finding\n\u2705 Local only; nothing leaves the machine, and it is a one-line config flip to undo\n\u274c Not ideal if you work on multiple client codebases where mixing context is a concern"
},
{
"label": "Keep project-scoped only",
"description": "\u2705 Strict isolation: reviews here only see this project's lessons\n\u2705 Safe default for multi-client work or shared machines\n\u274c Patterns you already learned on another repo will not be suggested here"
}
]
},
{
"question": "D3 \u2014 R2: Which review mode for the saved-views plan?\nProject/branch/task: gstack-plan-count-FwyQuk on main; reviewing PLAN.md \"Add saved project views\".\nELI10: The mode sets my posture for the rest of the review. Expansion pushes for the biggest version, Hold Scope stress-tests exactly what you wrote, Reduction strips to the smallest shippable core, and Selective holds your scope while offering a few add-ons one at a time for you to accept or decline.\nStakes if we pick wrong: too ambitious and a small feature balloons; too strict and we ship personal-only views when the goal (\"team members repeatedly recreate filters\") may really be a shared-view problem, forcing a second migration later.\nRecommendation: SELECTIVE EXPANSION because the plan is an added capability of ~8\u201310 files, but its member-only scoping is the one fact that could be wrong: every incumbent ships shared views too, and the table shape decides whether adding them later is a column or a rewrite. Selective lets you rule on that once without committing to a bigger build.\nNote: options differ in kind, not coverage \u2014 no completeness score.\nNet: how much of the review is spent challenging scope vs. hardening the scope you already chose.",
"header": "Review mode",
"multiSelect": false,
"options": [
{
"label": "SELECTIVE EXPANSION (recommended)",
"description": "\u2705 Keeps your four approach bullets as the baseline and hardens them with full rigor\n\u2705 Offers each expansion (shared views, default view, cleanup) as a separate add/defer/skip call\n\u274c A few more decision questions than Hold Scope before the deep review starts"
},
{
"label": "HOLD SCOPE",
"description": "\u2705 Maximum rigor on exactly what is written: error paths, edge cases, tests, observability\n\u2705 Fastest path to an implementation-ready plan with no scope questions\n\u274c Shared views and table-shape futureproofing get flagged, not offered; possible second migration later"
},
{
"label": "SCOPE EXPANSION",
"description": "\u2705 Designs the platonic saved-views feature: personal + shared, defaults, share links, cleanup\n\u2705 Best long-term architecture up front; no follow-up migrations\n\u274c Turns a ~10-file feature into a multi-surface build before the two-week pilot proves reuse"
},
{
"label": "SCOPE REDUCTION",
"description": "\u2705 Finds the smallest core that tests the pilot hypothesis (maybe create/list/apply only)\n\u2705 Lowest risk and fastest to the two-week reuse measurement\n\u274c Update/delete and picker polish get deferred; pilot may measure a clunky version of the feature"
}
]
}
],
"answered": false,
"failed": false
}
],
"assistantMessages": []
}
}
+22 -10
View File
@@ -149,7 +149,7 @@ function hasNativePostureProse(text: string, posture: RegExp): boolean {
/** Finish the selected native mode packet before waiting for its answer. */
export function ceoModeSubmissionInput(
visible: string, selected: NativePlanQuestionCall | undefined, targetMode: CeoMode,
transcript: PlanCountTranscript, submitted: Set<string>,
transcript: PlanCountTranscript, submitted: Set<string>, screenText = '',
): string | null {
if (!selected || selected.answered || selected.failed || !selected.sessionId || !selected.toolUseId ||
transcript.status !== 'ready' || selected.questions.length < 2 ||
@@ -161,16 +161,28 @@ export function ceoModeSubmissionInput(
const modeQuestions = selected.questions.filter(q => q.options.filter(o => modeTitle(o.label)).length >= 2);
if (modeQuestions.length !== 1 || findCeoModeOption(modeQuestions[0]!.options.map((o, i) =>
({index:i + 1, label:o.label})), targetMode) === null) return null;
const bar = posturePacketBar(visible);
if (!bar || !bar.answered.every(Boolean) || JSON.stringify(bar.headers) !== JSON.stringify(
selected.questions.map(q => q.header.trim().replace(/\s+/g, ' '))) ||
planCountSubmissionInput(visible) !== '\r') return null;
const rawBar = [...visible.matchAll(/←[^\r\n]+✔\s*Submit\s*→/g)].at(-1)!;
const preceding = visible.slice(0, rawBar.index);
if (/```|~~~|^\s*>|\b(?:example|quoted|source)[^:\n]*:\s*$/im.test(preceding)) return null;
const compact = (text: string) => text.replace(/\s+/g, '');
const panel = compact(visible.slice(rawBar.index! + rawBar[0].length)
.replace(/^[ \t]*[│┃] ?/gm, '').replace(/^[ \t]*[●⏺] ?/gm, ''));
const quotedContext = /```|~~~|^\s*>|\b(?:example|quoted|source)[^:\n]*:\s*$/im;
const bar = posturePacketBar(visible);
let review: string;
if (bar) {
if (!bar.answered.every(Boolean) || JSON.stringify(bar.headers) !== JSON.stringify(
selected.questions.map(q => q.header.trim().replace(/\s+/g, ' '))) ||
planCountSubmissionInput(visible) !== '\r') return null;
const rawBar = [...visible.matchAll(/←[^\r\n]+✔\s*Submit\s*→/g)].at(-1)!;
if (quotedContext.test(visible.slice(0, rawBar.index))) return null;
review = visible.slice(rawBar.index! + rawBar[0].length);
} else {
// A review taller than the terminal scrolls its tab bar and heading off
// the viewport (run 36606688266). The viewport must still end at the
// focused Submit prompt; the accumulated screen text then supplies the
// one complete review panel, authenticated below exactly as with a bar.
const heading = screenText.lastIndexOf('Review your answers');
if (heading < 0 || !compact(visible).endsWith(BARLESS_SUBMIT_END) ||
quotedContext.test(screenText.slice(0, heading).split('\n').slice(-3).join('\n'))) return null;
review = screenText.slice(heading);
}
const panel = compact(review.replace(/^[ \t]*[│┃] ?/gm, '').replace(/^[ \t]*[●⏺] ?/gm, ''));
// Authenticate the complete review panel against native questions and
// offered answers. An intended keypress or a selected-mode echo is not an ACK.
let prefixes = ['Reviewyouranswers'];
+1 -1
View File
@@ -260,7 +260,7 @@ describeE2E('/plan-ceo-review mode routing (gate)', () => {
}
const currentInput = await session.currentScreen();
capture('awaiting_posture', currentInput, transcript);
const modeSubmit = ceoModeSubmissionInput(currentInput, question.nativeCall, c.mode, transcript, submittedModePackets);
const modeSubmit = ceoModeSubmissionInput(currentInput, question.nativeCall, c.mode, transcript, submittedModePackets, session.visibleText());
if (modeSubmit !== null) { session.send(modeSubmit); continue; }
const pendingQuestion = readPendingQuestion(session.pendingQuestionFile, fixture.cwd,
session.hermeticConfigDir, selectionStartedAt, transcript);