From a27365bffe3ed2b3ea30067fe326bf890a9478cc Mon Sep 17 00:00:00 2001 From: garrytan Date: Wed, 30 Sep 2026 12:52:56 +0000 Subject: [PATCH] test: PTY harness handles clipped reviews and bundled setup tabs; AUQ judge uses structured output; design-consultation carve declines optional outside voices - ceo mode routing: a Submit review taller than the viewport, a setup tab bundled after the mode tab, and a clip through the mode question each hung or misread the run; the native answer is still verified after Submit. - judgeRecommendation requests a 1-5 enum schema; a malformed Haiku reply had scored substance 0 for a 4/5 brief. Judge failures now propagate. - carve section-loading for design-consultation declines the optional outside voices (a supported path) and treats DESIGN.md as the report; timeout unchanged. The Step 0E handoff defect is not fixed (0/15 samples across four wordings, none shipped) and is filed in TODOS. --- TODOS.md | 18 +- test/ceo-hold-posture-review.test.ts | 2 +- test/ceo-mode-option.test.ts | 138 ++++++++- test/ceo-mode-routing-fixture.test.ts | 1 + ...gn-consultation-section-completion.test.ts | 93 ++++++ test/fixtures/ceo-mode-bundled-tab-local.json | 84 ++++++ .../ceo-mode-clipped-mode-question-local.json | 84 ++++++ .../ceo-mode-clipped-review-local.json | 71 +++++ .../design-consultation-section-design-md.md | 279 ++++++++++++++++++ test/helpers/auq-sdk-capture.ts | 11 +- test/helpers/carve-guards.ts | 2 +- test/helpers/carve-section-case.ts | 8 +- test/helpers/ceo-mode-option.ts | 76 +++++ test/helpers/llm-judge.ts | 12 +- test/llm-judge-frontier.test.ts | 32 +- test/skill-e2e-plan-ceo-mode-routing.test.ts | 10 +- 16 files changed, 893 insertions(+), 28 deletions(-) create mode 100644 test/design-consultation-section-completion.test.ts create mode 100644 test/fixtures/ceo-mode-bundled-tab-local.json create mode 100644 test/fixtures/ceo-mode-clipped-mode-question-local.json create mode 100644 test/fixtures/ceo-mode-clipped-review-local.json create mode 100644 test/fixtures/design-consultation-section-design-md.md diff --git a/TODOS.md b/TODOS.md index 00ace9d87..90c13f577 100644 --- a/TODOS.md +++ b/TODOS.md @@ -13,17 +13,13 @@ 2.1.251 in every recent run) and the HOLD SCOPE routing case when its next brief happens not to name the mode (see the handoff item below). Effort M each. -- **`/plan-ceo-review` skips its Step 0E mode handoff** — in 4 of 4 asked-mode - samples (census 36633323521 plus three local runs) the model went straight - to tools after the mode answer without the required `Mode: ; approved - decisions: …` chat. The routing case still passes on other posture text; a - wording change moving the handoff ahead of the question log did not change - the behavior in two paid runs, so it was not shipped. Effort M. -- **`/ship` design-lite probe wording** — `scripts/resolvers/design.ts` still - says "Probe for a design detector the user installed", the wording that let - `/review` skip its probe in 5 of 6 captured trials before this release made it - mandatory. The ship union ratio is at 1.3966 of 1.397, so the same sentence - needs a trim elsewhere first. Effort S. +- **`/plan-ceo-review` skips its Step 0E mode handoff** — 0 of 15 answered + samples sent the required `Mode: ; approved decisions: …` chat after + the mode answer, across four wording repairs (none shipped). The model writes + the handoff in its reasoning and later says it was "sent above". A prose fix + won't reach it; this needs a mechanism outside the prompt (a hook or a + tool-result gate). The HOLD SCOPE routing case fails whenever the handoff is + skipped and nothing else names the posture in time. Effort M. - **Let pass-rate history decide the rest** — every census on this branch had a different handful of single-trial reds. Once `eval:pass-rates` has 10 weekly trials per case, apply the CASE_QUARANTINE entry rule instead of chasing one diff --git a/test/ceo-hold-posture-review.test.ts b/test/ceo-hold-posture-review.test.ts index ccc814e15..11accf0cb 100644 --- a/test/ceo-hold-posture-review.test.ts +++ b/test/ceo-hold-posture-review.test.ts @@ -243,7 +243,7 @@ async function registered(scenario:'accept'|'uncertain'|'missing source'|'missin navigateToModeAskUserQuestion:async()=>({modeIndex:3,visibleAtMode:'captured mode',question:{nativeCall:mode(f)}}), planCountQuestionInput:(_v:string,q:any)=>q.nativeCall.toolUseId===modeId?'3':'1',selectPtyNumberedOption:async()=>{throw Error('unexpected legacy key');}, hasNativePostAnswerCeoPosture:scenario==='lexical pass'||scenario==='expansion'?()=>true:hasNativePostAnswerCeoPosture, - ceoModeSubmissionInput:()=>null,ceoExpansionPacingReady:()=>false,ceoExpansionPacingChoice:()=>null,holdDeferKeepIndex, + ceoModeSubmissionInput:()=>null,ceoModePacketTabAnswer:()=>null,ceoExpansionPacingReady:()=>false,ceoExpansionPacingChoice:()=>null,holdDeferKeepIndex, nextCeoPostureContinuation:(_a:any,_b:any,_c:any,_d:any,_e:any,continued:boolean)=>continued?null:'question', capturePlanCountQuestion:()=>({nativeCall:pending}),isPlanReadyVisible:()=>false,isNumberedOptionListVisible:()=>false, buildCeoHoldPostureReview,evaluateCeoHoldPostureReview:async(review:PlanReviewDecisionInput)=>{ diff --git a/test/ceo-mode-option.test.ts b/test/ceo-mode-option.test.ts index 20146dc3c..c1031727c 100644 --- a/test/ceo-mode-option.test.ts +++ b/test/ceo-mode-option.test.ts @@ -11,12 +11,15 @@ import captured_ceo_hold_posture_ag from './fixtures/ceo-hold-posture-ag.json'; import retainedPreservationCaptures_ceo_hold_posture_ag from './fixtures/ceo-hold-preservation-f359.json'; import captured_ceo_mode_colon_at from './fixtures/ceo-mode-colon-at.json'; import scrolledReview from './fixtures/ceo-mode-scrolled-review-36606688266.json'; +import clippedReview from './fixtures/ceo-mode-clipped-review-local.json'; +import bundledTab from './fixtures/ceo-mode-bundled-tab-local.json'; +import clippedMode from './fixtures/ceo-mode-clipped-mode-question-local.json'; import fs_ceo_mode_full_ad from 'node:fs'; import os_ceo_mode_full_ad from 'node:os'; import path_ceo_mode_full_ad from 'node:path'; import { ceoExpansionPacingChoice } from './helpers/ceo-mode-option'; import { ceoExpansionPacingReady } from './helpers/ceo-mode-option'; -import { ceoModeSubmissionInput } from './helpers/ceo-mode-option'; +import { ceoModePacketTabAnswer, ceoModeSubmissionInput } from './helpers/ceo-mode-option'; import { capturePlanCountQuestion } from './helpers/claude-pty-runner'; import { planCountPrerequisitePick } from './helpers/claude-pty-runner'; import { isNumberedOptionListVisible } from './helpers/claude-pty-runner'; @@ -1628,7 +1631,7 @@ test.each(['acknowledged pacing','missing pacing ACK'])('actual paid posture loo expect(start).toBeGreaterThan(0);expect(end).toBeGreaterThan(start); const loop=source.slice(start,end+" outcome = 'posture_confirmed';".length); const keys=['Bun','Date','c','session','sincePick','selectionStartedAt','question','fixture','capture','readPlanCountTranscript', - 'readPendingQuestion','hasNativePostAnswerCeoPosture','ceoModeSubmissionInput','ceoExpansionPacingReady','ceoExpansionPacingChoice', + 'readPendingQuestion','hasNativePostAnswerCeoPosture','ceoModeSubmissionInput','ceoModePacketTabAnswer','ceoExpansionPacingReady','ceoExpansionPacingChoice', 'nextCeoPostureContinuation','capturePlanCountQuestion','planCountQuestionInput','selectPtyNumberedOption','isPlanReadyVisible','isNumberedOptionListVisible', 'EXPANSION_PACING_CALLS','modeIndex','artifacts','visibleAtMode','postureSource']; const compiled=new Bun.Transpiler({loader:'ts'}).transformSync(`async function run(b){const {${keys.join(',')}}=b;let outcome;${loop};return {outcome,continuedQuestion,pacingCalls};}`); @@ -1655,7 +1658,7 @@ test.each(['acknowledged pacing','missing pacing ACK'])('actual paid posture loo }; const bindings={Bun:{sleep:async(ms:number)=>{clock+=ms;}},Date:{now:()=>clock},c:{mode:'SCOPE EXPANSION',postureRe:pattern},session,sincePick:0, selectionStartedAt:f.selectedAt,question:{nativeCall:f.mode},fixture:{cwd:'fixture-root'},capture:(state:string)=>snapshots.push(state),readPlanCountTranscript, - readPendingQuestion:()=>undefined,hasNativePostAnswerCeoPosture,ceoModeSubmissionInput,ceoExpansionPacingReady,ceoExpansionPacingChoice,nextCeoPostureContinuation, + readPendingQuestion:()=>undefined,hasNativePostAnswerCeoPosture,ceoModeSubmissionInput,ceoModePacketTabAnswer,ceoExpansionPacingReady,ceoExpansionPacingChoice,nextCeoPostureContinuation, capturePlanCountQuestion,planCountQuestionInput,selectPtyNumberedOption:async(s:any,index:number)=>s.send(String(index)),isPlanReadyVisible,isNumberedOptionListVisible, EXPANSION_PACING_CALLS:1,modeIndex:2,artifacts:{},visibleAtMode:'captured mode menu', postureSource:{path:path.join('fixture-root','PLAN.md'),content:plan}}; @@ -1894,6 +1897,135 @@ describe('mode submission when the review panel scrolls past the viewport', () = }); }); +describe('a setup tab bundled after the mode tab', () => { + const transcript = bundledTab.transcript as unknown as PlanCountTranscript; + const call = transcript.calls[1] as NativePlanQuestionCall; + const answer = (screen = bundledTab.screen, selected: NativePlanQuestionCall = call, native = transcript, seen = new Set()) => + ceoModePacketTabAnswer(screen, selected, native, seen); + + test('the captured packet asks the mode first and Learnings second', () => { + expect(call.questions.map(q => q.header)).toEqual(['Review mode', 'Learnings']); + expect(bundledTab.screen).toContain('☒ Review mode ☐ Learnings ✔ Submit'); + }); + + test('the answered mode tab lets the harness answer the Learnings tab once with option 1', () => { + const seen = new Set(); + const first = answer(bundledTab.screen, call, transcript, seen); + expect(first?.index).toBe(1); + expect(first?.question.nativeQuestionIndex).toBe(1); + expect(answer(bundledTab.screen, call, transcript, seen)).toBeNull(); + }); + + test('an unanswered mode tab is left for the mode selection', () => { + expect(answer(bundledTab.screen.replace('☒ Review mode', '☐ Review mode'))).toBeNull(); + }); + + test('an already answered setup tab is not answered again', () => { + expect(answer(bundledTab.screen.replace('☐ Learnings', '☒ Learnings'))).toBeNull(); + }); + + test('a tab bar naming other questions does not belong to this packet', () => { + expect(answer(bundledTab.screen.replace('☐ Learnings', '☐ Deploy'))).toBeNull(); + }); + + test('an answered, foreign or changed native call gets no input', () => { + expect(answer(bundledTab.screen, { ...call, answered: true })).toBeNull(); + expect(answer(bundledTab.screen, { ...call, toolUseId: 'foreign' })).toBeNull(); + const changed = structuredClone(call); + changed.questions[1]!.options[1]!.label = 'Upload learnings'; + expect(answer(bundledTab.screen, changed, { ...transcript, calls: [transcript.calls[0]!, changed] })).toBeNull(); + }); + + test('a setup tab before the mode tab stays with navigation', () => { + const reordered = structuredClone(call); + reordered.questions.reverse(); + const screen = bundledTab.screen.replace('☒ Review mode ☐ Learnings', '☐ Learnings ☒ Review mode'); + expect(answer(screen, reordered, { ...transcript, calls: [transcript.calls[0]!, reordered] })).toBeNull(); + }); +}); + +describe('mode submission when the clip cuts through the mode question itself', () => { + const transcript = clippedMode.transcript as unknown as PlanCountTranscript; + const call = transcript.calls[1] as NativePlanQuestionCall; + const submit = (screen: string, mode: 'HOLD SCOPE' | 'SCOPE EXPANSION' = 'SCOPE EXPANSION', selected = call) => + ceoModeSubmissionInput(screen, selected, mode, transcript, new Set(), screen); + + test('the captured viewport starts inside the mode question and still shows its answer', () => { + expect(call.questions.map(q => q.header)).toEqual(['Review mode', 'Learnings']); + expect(clippedMode.screen).not.toContain('Review your answers'); + expect(clippedMode.screen).not.toContain('Which review mode'); + expect(clippedMode.screen).toMatch(/→ SCOPE EXPANSION[\s\S]*→ Enable cross-project learnings \(recommended\)\s+Ready to submit/); + }); + + test('the visible target answer and a long native tail submit once', () => { + const seen = new Set(); + expect(ceoModeSubmissionInput(clippedMode.screen, call, 'SCOPE EXPANSION', transcript, seen, clippedMode.screen)).toBe('\r'); + expect(ceoModeSubmissionInput(clippedMode.screen, call, 'SCOPE EXPANSION', transcript, seen, clippedMode.screen)).toBeNull(); + }); + + test('another target mode is not acknowledged', () => { + expect(submit(clippedMode.screen, 'HOLD SCOPE')).toBeNull(); + }); + + test('a short remnant of the mode question cannot identify it', () => { + const cut = clippedMode.screen.lastIndexOf('\n', clippedMode.screen.indexOf('→ SCOPE EXPANSION') - 2); + expect(submit(clippedMode.screen.slice(cut + 1))).toBeNull(); + }); + + test('an altered mode question tail is rejected', () => { + expect(submit(clippedMode.screen.replace('Avoids a later schema migration', 'Avoids a later deploy'))).toBeNull(); + }); +}); + +describe('mode submission when the review panel is clipped before its heading renders', () => { + const transcript = clippedReview.transcript as unknown as PlanCountTranscript; + const call = transcript.calls[0] as NativePlanQuestionCall; + const submit = (screen: string, mode: 'HOLD SCOPE' | 'SCOPE EXPANSION' = 'HOLD SCOPE', selected = call) => + ceoModeSubmissionInput(screen, selected, mode, transcript, new Set(), screen); + + test('the captured review has no heading or tab bar and truncates the mode question', () => { + expect(clippedReview.screen).not.toContain('Review your answers'); + expect(clippedReview.screen).not.toMatch(/←[^\r\n]+✔\s*Submit\s*→/); + expect(clippedReview.screen).toMatch(/flagged as …\s+→ HOLD SCOPE\s+Ready to submit your answers\?/); + expect(call.questions.map(q => q.header)).toEqual(['Routing', 'Learnings', 'Review mode']); + }); + + test('the clipped review submits the selected mode once', () => { + const seen = new Set(); + expect(ceoModeSubmissionInput(clippedReview.screen, call, 'HOLD SCOPE', transcript, seen, clippedReview.screen)).toBe('\r'); + expect(ceoModeSubmissionInput(clippedReview.screen, call, 'HOLD SCOPE', transcript, seen, clippedReview.screen)).toBeNull(); + }); + + test('without accumulated screen text ending at the same prompt the clipped route cannot submit', () => { + expect(ceoModeSubmissionInput(clippedReview.screen, call, 'HOLD SCOPE', transcript, new Set(), '')).toBeNull(); + expect(ceoModeSubmissionInput(clippedReview.screen, call, 'HOLD SCOPE', transcript, new Set(), `${clippedReview.screen}\nMore`)).toBeNull(); + }); + + test('a clipped review showing another mode does not acknowledge the target', () => { + expect(submit(clippedReview.screen, 'SCOPE EXPANSION')).toBeNull(); + }); + + for (const [name, change] of [ + ['an altered mode question', (text: string) => text.replace('D3 — MODE: Which review mode', 'D3 — MODE: Which deploy mode')], + ['an altered truncated tail', (text: string) => text.replace('get flagged as …', 'get deleted as …')], + ['a mode question truncated too early', (text: string) => text.replace(/│ ● D3 — MODE:[\s\S]*?→ HOLD SCOPE/, '│ ● D3 — MODE: Which review mode for the saved-views plan?…\n → HOLD SCOPE')], + ['an answer no option offers', (text: string) => text.replace('→ Enable cross-project (recommended)', '→ Upload learnings')], + ['a clip that hides the mode answer', (text: string) => text.slice(text.indexOf('Ready to submit'))], + ['a visible tab bar', (text: string) => `← ☒ Routing ☒ Learnings ☐ Review mode ✔ Submit →\n${text}`], + ['output after the prompt', (text: string) => `${text}\nMore text`], + ['an unfocused Submit prompt', (text: string) => text.replace('❯ 1. Submit answers', ' 1. Submit answers\n❯ 2. Cancel')], + ] as const) test(`the clipped route rejects ${name}`, () => { + expect(submit(change(clippedReview.screen))).toBeNull(); + }); + + test('an answered or changed native call cannot be submitted', () => { + expect(submit(clippedReview.screen, 'HOLD SCOPE', { ...call, answered: true })).toBeNull(); + const other = structuredClone(call); + other.questions[2]!.question = other.questions[2]!.question.replace('Which review mode', 'Which deploy mode'); + expect(submit(clippedReview.screen, 'HOLD SCOPE', other)).toBeNull(); + }); +}); + describe('HOLD SCOPE defer/keep menu (census 36626737820: "Defer update to TODOS.md" was answered as the rigor decision)', () => { const call = (labels: string[], extra: Record = {}) => ({ sessionId: 's', toolUseId: 't', answered: false, failed: false, diff --git a/test/ceo-mode-routing-fixture.test.ts b/test/ceo-mode-routing-fixture.test.ts index c90d9d375..db82dc5ea 100644 --- a/test/ceo-mode-routing-fixture.test.ts +++ b/test/ceo-mode-routing-fixture.test.ts @@ -52,6 +52,7 @@ mock.module(path.join(root,'test/helpers/ceo-mode-option.ts'),()=>({ ceoExpansionPacingChoice:()=>{current.pacingChoices++;return scenario.startsWith('pacing')&&(!current.pacingSent||scenario==='pacing-repeated')?{call:{questions:[]},index:scenario==='pacing-unsupported'?0:1}:null;}, ceoExpansionPacingReady:()=>{current.pacingChecks++;return (scenario==='pacing'||scenario==='pacing-repeated')&¤t.pacingChecks>=3;}, ceoModeSubmissionInput:()=>scenario==='mode-submit'&&!current.submitted?'\\r':null, + ceoModePacketTabAnswer:()=>null, holdDeferKeepIndex:()=>null, nextCeoModeNavigation:(_visible,target)=>{ if(scenario==='navigation')throw new Error('fixture navigation failed'); diff --git a/test/design-consultation-section-completion.test.ts b/test/design-consultation-section-completion.test.ts new file mode 100644 index 000000000..1439056f0 --- /dev/null +++ b/test/design-consultation-section-completion.test.ts @@ -0,0 +1,93 @@ +import { expect, test } from 'bun:test'; +import * as fs from 'node:fs'; +import * as os from 'node:os'; +import * as path from 'node:path'; + +const ROOT = path.resolve(import.meta.dir, '..'); +// DESIGN.md content from the Write call of the census 36641820398 slice 8 +// design-consultation capture. That run Read its section at 11s and wrote +// DESIGN.md and CLAUDE.md, then timed out composing the duplicate REPORT.md +// the generic fixture requested. +const designMd = fs.readFileSync(path.join(import.meta.dir, 'fixtures/design-consultation-section-design-md.md'), 'utf8'); +const genericReport = '# Design report\n' + 'The design review summary is complete. '.repeat(8); + +interface Fixture { + output?: string; + file?: 'DESIGN.md' | 'REPORT.md' | null; + exitReason?: string; + missingRead?: boolean; +} + +// Run the actual paid registration and capture helper in an isolated free +// child; only the session-runner/provider boundary is replaced. +function exercise(fixture: Fixture = {}) { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'design-consultation-completion-')); + const script = path.join(dir, 'capture.test.ts'); + const facts = path.join(dir, 'facts.json'); + const input = { output: designMd, file: 'DESIGN.md', exitReason: 'success', missingRead: false, ...fixture }; + fs.writeFileSync(script, ` +import { expect, mock } from 'bun:test'; +import * as fs from 'node:fs'; +import * as path from 'node:path'; +import { CARVE_GUARDS } from ${JSON.stringify(path.join(ROOT, 'test/helpers/carve-guards.ts'))}; +const input = ${JSON.stringify(input)}; +const guard = CARVE_GUARDS['design-consultation']; +mock.module(${JSON.stringify(path.join(ROOT, 'test/helpers/session-runner.ts'))}, () => ({ + runSkillTest: async opts => { + fs.writeFileSync(${JSON.stringify(facts)}, JSON.stringify({ prompt: opts.prompt, timeout: opts.timeout, allowedTools: opts.allowedTools })); + if (input.file) fs.writeFileSync(path.join(opts.workingDirectory, input.file), input.output); + return { + exitReason: input.exitReason, output: 'Wrote DESIGN.md and CLAUDE.md.', + toolCalls: input.missingRead ? [] : guard.requiredReads.map(section => ({ + tool: 'Read', input: { file_path: path.join(opts.workingDirectory, 'design-consultation', 'sections', section) }, + })), transcript: [], + }; + }, +})); +const { registerCarveSectionCase } = await import(${JSON.stringify(path.join(ROOT, 'test/helpers/carve-section-case.ts'))}); +registerCarveSectionCase('design-consultation'); +`); + try { + const child = Bun.spawnSync([process.execPath, 'test', script], { + cwd: ROOT, + env: { + PATH: process.env.PATH ?? '', HOME: dir, TMPDIR: dir, TEMP: dir, TMP: dir, + ...(process.env.SystemRoot ? { SystemRoot: process.env.SystemRoot } : {}), + }, + timeout: 10_000, + }); + expect(fs.existsSync(facts), child.stderr.toString()).toBe(true); + expect(child.signalCode ?? null).toBeNull(); + return { code: child.exitCode, output: child.stdout.toString() + child.stderr.toString(), + facts: JSON.parse(fs.readFileSync(facts, 'utf8')) as { prompt: string; timeout: number; allowedTools: string[] } }; + } finally { + fs.rmSync(dir, { recursive: true, force: true }); + } +} + +test('the captured DESIGN.md is the completed output; no duplicate report is requested', () => { + expect(designMd.split('\n')[1]).toBe('# gstack: design-md-format=spec'); + const result = exercise(); + expect(result.code, result.output).toBe(0); + expect(result.facts.timeout).toBe(480_000); + expect(result.facts.prompt).toContain('declined the optional outside design voices'); + expect(result.facts.prompt).toMatch(/write the skill's final output[^\n]*DESIGN\.md/); + expect(result.facts.prompt).not.toContain('REPORT.md'); +}, 20_000); + +test('the census timeout still fails even after DESIGN.md was written', () => { + expect(exercise({ exitReason: 'timeout' }).code).not.toBe(0); +}, 20_000); + +test('a generic report without the DESIGN.md format marker does not count as completion', () => { + expect(exercise({ output: genericReport }).code).not.toBe(0); + expect(exercise({ output: genericReport, file: 'REPORT.md' }).code).not.toBe(0); +}); + +test('a terminal-only claim without writing DESIGN.md fails', () => { + expect(exercise({ file: null }).code).not.toBe(0); +}, 20_000); + +test('skipping the section Read fails', () => { + expect(exercise({ missingRead: true }).code).not.toBe(0); +}, 20_000); diff --git a/test/fixtures/ceo-mode-bundled-tab-local.json b/test/fixtures/ceo-mode-bundled-tab-local.json new file mode 100644 index 000000000..32a83f729 --- /dev/null +++ b/test/fixtures/ceo-mode-bundled-tab-local.json @@ -0,0 +1,84 @@ +{ + "source": "Local paid proof run 2026-09-30 (mode routing, HOLD SCOPE case, Claude Code 2.1.251): the model bundled the Learnings setup question after the mode question in one native call. The harness selected HOLD SCOPE on the mode tab, then never answered the Learnings tab, so Submit was unreachable and the case ran out its posture budget.", + "screen": "Planning: /tmp/gstack-hermetic-fixture/with-skills/.claude/plans/swirling-mixing-spindle.md\n────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────\n← ☒ Review mode ☐ Learnings ✔ Submit →\n\n│ D3 — One-time gstack setup: search learnings across your other projects on this machine?\n│ Project/branch/task: gstack-plan-count on main; gstack config, not the plan.\n│ ELI10: gstack stores small lessons per project (\"this repo's tests need X\"). It can also look at lessons from your\n│ other local projects when reviewing this one. Everything stays on this machine. Right now zero learnings are stored,\n│ so this only affects future sessions.\n│ Stakes if we pick wrong: Cross-project on with client codebases could mix contexts; off means each project learns\n│ alone.\n│ Recommendation: Enable because it is local-only and you appear to be a solo operator on this machine.\n│ Note: options differ in kind, not coverage — no completeness score.\n│ Pros / cons:\n│ A) Enable cross-project learnings (recommended)\n│ ✅ Patterns learned in one repo (tooling quirks, review habits) carry into the next\n│ ✅ Local only; nothing leaves the machine; toggle back off with one command\n│ ❌ Not ideal if you review multiple client codebases that must stay isolated\n│ B) Keep learnings project-scoped\n│ ✅ Hard isolation between projects on this machine\n│ ✅ Zero risk of a lesson from repo A being misapplied to repo B\n│ ❌ Every project starts cold; the same tooling lesson gets relearned per repo\n│ Net: Compounding across repos vs. strict isolation.\n\n❯ 1. Enable cross-project (recommended)\n ✅ Lessons carry between your local repos. ✅ Local only, reversible with one command. ❌ Less ideal if client\n codebases must stay isolated.\n 2. Project-scoped only\n ✅ Hard isolation between projects. ✅ No misapplied lessons across repos. ❌ Every project starts cold.\n 3. Type something.\n────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────\n 4. Chat about this\n\nEnter to select · Tab/Arrow keys to navigate · Esc to cancel\n\n\n\n\n\n", + "transcript": { + "status": "ready", + "calls": [ + { + "sessionId": "bca7f78c-bf20-4ba2-ac64-c20ef681e342", + "toolUseId": "toolu_01Cyhcx3LZpvA5aqV239wmhs", + "questions": [ + { + "question": "gstack works best when your project's CLAUDE.md includes skill routing rules. Add them?\n\nD1 — Add gstack skill routing rules to CLAUDE.md?\nProject/branch/task: gstack-plan-count on main, starting /plan-ceo-review of PLAN.md (saved project views).\nELI10: gstack has many skills (/investigate, /ship, /qa...). A short routing section in CLAUDE.md tells Claude which skill to reach for when you describe a task in plain words, so you don't have to remember slash names. This is a one-time setup, separate from the plan review.\nStakes if we pick wrong: Without it, skills only fire when you type them explicitly; with it, a few lines get appended to CLAUDE.md and committed.\nRecommendation: A because routing is cheap and makes the rest of gstack discoverable.\nNote: options differ in kind, not coverage — no completeness score.\nPros / cons:\nA) Add routing rules (recommended)\n ✅ Plain-language requests like \"find this bug\" auto-route to the right skill\n ✅ One-time append and commit; nothing else in the repo changes\n ❌ Adds ~20 lines to CLAUDE.md; commit happens after plan mode ends since edits are frozen now\nB) No thanks, invoke manually\n ✅ CLAUDE.md stays exactly as-is for this fixture repo\n ✅ Re-enable any time with one config command\n ❌ Skills stay invisible unless you type the slash name\nNet: Discoverability vs. an untouched CLAUDE.md.", + "header": "Routing", + "multiSelect": false, + "options": [ + { + "label": "Add routing rules (recommended)", + "description": "✅ Plain-language requests auto-route to the matching gstack skill. ✅ One-time append and commit. ❌ Adds ~20 lines to CLAUDE.md; the write waits until plan mode ends." + }, + { + "label": "No thanks, invoke manually", + "description": "✅ CLAUDE.md stays untouched. ✅ Re-enable later with one config command. ❌ Skills only fire when typed explicitly." + } + ] + } + ], + "answered": true, + "failed": false, + "answers": { + "gstack works best when your project's CLAUDE.md includes skill routing rules. Add them?\n\nD1 — Add gstack skill routing rules to CLAUDE.md?\nProject/branch/task: gstack-plan-count on main, starting /plan-ceo-review of PLAN.md (saved project views).\nELI10: gstack has many skills (/investigate, /ship, /qa...). A short routing section in CLAUDE.md tells Claude which skill to reach for when you describe a task in plain words, so you don't have to remember slash names. This is a one-time setup, separate from the plan review.\nStakes if we pick wrong: Without it, skills only fire when you type them explicitly; with it, a few lines get appended to CLAUDE.md and committed.\nRecommendation: A because routing is cheap and makes the rest of gstack discoverable.\nNote: options differ in kind, not coverage — no completeness score.\nPros / cons:\nA) Add routing rules (recommended)\n ✅ Plain-language requests like \"find this bug\" auto-route to the right skill\n ✅ One-time append and commit; nothing else in the repo changes\n ❌ Adds ~20 lines to CLAUDE.md; commit happens after plan mode ends since edits are frozen now\nB) No thanks, invoke manually\n ✅ CLAUDE.md stays exactly as-is for this fixture repo\n ✅ Re-enable any time with one config command\n ❌ Skills stay invisible unless you type the slash name\nNet: Discoverability vs. an untouched CLAUDE.md.": "Add routing rules (recommended)" + }, + "unansweredQuestionIndices": [], + "answeredAt": "2026-09-30T11:51:49.397Z" + }, + { + "sessionId": "bca7f78c-bf20-4ba2-ac64-c20ef681e342", + "toolUseId": "toolu_01SzGzrL9JDP3jbR3t8unp3R", + "questions": [ + { + "question": "D2 — Which review mode for the saved project views plan?\nProject/branch/task: gstack-plan-count on main, reviewing PLAN.md (saved views: table + CRUD + picker).\nELI10: The mode sets my posture for the rest of the review. Expansion means I pitch bigger versions of the feature. Selective means I harden what you wrote and offer add-ons one at a time for you to accept or reject. Hold means I only stress-test what is there. Reduction means I look for what to cut. The plan is an added capability on an existing task list, about ~12 changed files (estimate; the app code is not in this repo).\nStakes if we pick wrong: Too big a mode and you get pitched a views platform you did not ask for; too small and the schema ships without a visibility column, which turns team-shared views into a migration plus permission rework in six months.\nRecommendation: SELECTIVE EXPANSION because the plan's premise (\"team members repeatedly recreate filters\") likely covers shared filters too, and a one-column decision now is much cheaper than a migration later, but that is your call per item, not a blanket expansion.\nNote: options differ in kind, not coverage — no completeness score.\nPros / cons:\nA) SELECTIVE EXPANSION (recommended)\n ✅ Hardens the current scope first (HOLD checks run in full), then offers each add-on separately\n ✅ Surfaces the visibility-column and filter-versioning forks while they are still one-column decisions\n ❌ A few more questions than HOLD; each add-on is its own accept/defer/skip\nB) HOLD SCOPE\n ✅ Fastest path: maximum rigor on exactly the four approach bullets you wrote\n ✅ No scope debate; every finding is about failure paths, tests, observability\n ❌ Leaves the shared-views fork unexamined; you may pay for it as a migration later\nC) SCOPE EXPANSION\n ✅ Aims at the 12-month ideal: shared views, defaults, dirty indicators, shareable links\n ✅ Best if the pilot goal is to make views the primary way people navigate tasks\n ❌ ~12 files becomes 20+; the two-week pilot metric gets muddied by feature count\nD) SCOPE REDUCTION\n ✅ Strips to the minimum that proves reuse in the pilot (e.g. create + list + apply only)\n ✅ Smallest diff, fastest to ship\n ❌ Update/delete are cheap and their absence makes the pilot data worse (stale views nobody can remove)\nNet: How much of the 12-month ideal you want to decide now versus after the pilot.", + "header": "Review mode", + "multiSelect": false, + "options": [ + { + "label": "SELECTIVE EXPANSION (recommended)", + "description": "✅ Hardens current scope first, then offers each add-on separately for accept/defer/skip. ✅ Surfaces the visibility-column and filter-versioning forks while cheap. ❌ A few more questions than HOLD." + }, + { + "label": "HOLD SCOPE", + "description": "✅ Maximum rigor on exactly the four approach bullets; fastest. ✅ Findings focus on failure paths, tests, observability. ❌ Shared-views fork stays unexamined; possible migration later." + }, + { + "label": "SCOPE EXPANSION", + "description": "✅ Aims at the 12-month ideal: shared views, defaults, dirty indicator, shareable links. ✅ Best if views become the primary navigation. ❌ ~12 files becomes 20+; pilot metric gets muddied." + }, + { + "label": "SCOPE REDUCTION", + "description": "✅ Minimum that proves reuse: create + list + apply. ✅ Smallest diff. ❌ Dropping update/delete leaves stale views nobody can remove, hurting pilot data." + } + ] + }, + { + "question": "D3 — One-time gstack setup: search learnings across your other projects on this machine?\nProject/branch/task: gstack-plan-count on main; gstack config, not the plan.\nELI10: gstack stores small lessons per project (\"this repo's tests need X\"). It can also look at lessons from your other local projects when reviewing this one. Everything stays on this machine. Right now zero learnings are stored, so this only affects future sessions.\nStakes if we pick wrong: Cross-project on with client codebases could mix contexts; off means each project learns alone.\nRecommendation: Enable because it is local-only and you appear to be a solo operator on this machine.\nNote: options differ in kind, not coverage — no completeness score.\nPros / cons:\nA) Enable cross-project learnings (recommended)\n ✅ Patterns learned in one repo (tooling quirks, review habits) carry into the next\n ✅ Local only; nothing leaves the machine; toggle back off with one command\n ❌ Not ideal if you review multiple client codebases that must stay isolated\nB) Keep learnings project-scoped\n ✅ Hard isolation between projects on this machine\n ✅ Zero risk of a lesson from repo A being misapplied to repo B\n ❌ Every project starts cold; the same tooling lesson gets relearned per repo\nNet: Compounding across repos vs. strict isolation.", + "header": "Learnings", + "multiSelect": false, + "options": [ + { + "label": "Enable cross-project (recommended)", + "description": "✅ Lessons carry between your local repos. ✅ Local only, reversible with one command. ❌ Less ideal if client codebases must stay isolated." + }, + { + "label": "Project-scoped only", + "description": "✅ Hard isolation between projects. ✅ No misapplied lessons across repos. ❌ Every project starts cold." + } + ] + } + ], + "answered": false, + "failed": false + } + ], + "assistantMessages": [] + } +} diff --git a/test/fixtures/ceo-mode-clipped-mode-question-local.json b/test/fixtures/ceo-mode-clipped-mode-question-local.json new file mode 100644 index 000000000..8f5c4eed8 --- /dev/null +++ b/test/fixtures/ceo-mode-clipped-mode-question-local.json @@ -0,0 +1,84 @@ +{ + "source": "Local paid proof run 2026-09-30 (mode routing, SCOPE EXPANSION case, Claude Code 2.1.251): one native call asked the mode then Learnings. After both tabs were answered, the Submit review was taller than the viewport; the mode question start and the review heading never rendered, and the lossy accumulated text could not authenticate them. The harness never submitted.", + "screen": " │ A) SELECTIVE EXPANSION (recommended)\n │ ✅ Keeps table + CRUD + picker fixed while you decide each add-on (shared views, default view, share links)\n │ individually\n │ ✅ Still runs the full HOLD rigor: error map, stale-filter failure modes, tests, observability\n │ ❌ More questions than HOLD; each add-on is a separate accept/defer/skip decision\n │ B) SCOPE EXPANSION\n │ ✅ Designs views as first-class objects from day one: personal, shared, defaults, automation targets\n │ ✅ Avoids a later schema migration if shared views are inevitable\n │ ❌ Likely pushes past 15 files and past the two-week pilot the plan budgets for\n │ C) HOLD SCOPE\n │ ✅ Fastest path to the pilot; maximum rigor on exactly the four bullets you wrote\n │ ✅ Cleanest read on whether members reuse views at all before adding surface area\n │ ❌ Locks in per-member-only schema; shared views become a migration if the pilot asks for them\n │ D) SCOPE REDUCTION\n │ ✅ Would strip to the smallest testable core (save + re…\n → SCOPE EXPANSION\n │ ● D3 — Enable cross-project learnings search?\n │ Project/branch/task: gstack-plan-count on main; one-time gstack config, not a plan decision.\n │ ELI10: gstack keeps notes from past sessions (quirks, pitfalls, patterns). With this on, reviews on this machine\n │ can also read notes written while working on your other projects. Everything stays local.\n │ Stakes if we pick wrong: Off means slower compounding across projects. On means notes from one client's codebase\n │ could surface while working on another's.\n │ Recommendation: Enable because this looks like a solo setup and local-only recall is low risk.\n │ Note: options differ in kind, not coverage — no completeness score.\n │ Pros / cons:\n │ A) Enable cross-project learnings (recommended)\n │ ✅ Patterns learned in one repo carry into reviews on the next one automatically\n │ ✅ Stays on this machine; nothing is uploaded or shared\n │ ❌ Not ideal if you juggle multiple client codebases that must stay separate\n │ B) Keep learnings project-scoped\n │ ✅ Hard wall between projects; no cross-contamination of notes\n │ ✅ Simplest mental model for multi-client work\n │ ❌ Each project relearns the same lessons from scratch\n │ Net: Compounding across projects versus strict per-project isolation.\n → Enable cross-project learnings (recommended)\n\nReady to submit your answers?\n \n❯ 1. Submit answers\n 2. Cancel", + "transcript": { + "status": "ready", + "calls": [ + { + "sessionId": "2db5ee57-7fe2-4d95-8b3f-8ad9334707f2", + "toolUseId": "toolu_0177nMqQ7xt94y33cWHGb1qe", + "questions": [ + { + "question": "gstack works best when your project's CLAUDE.md includes skill routing rules. Add them?\n\nD1 — Add gstack skill routing rules to CLAUDE.md?\nProject/branch/task: gstack-plan-count on main, starting /plan-ceo-review of PLAN.md (saved project views).\nELI10: gstack ships a dozen skills (review, ship, investigate, etc.). A short routing table in CLAUDE.md tells Claude which skill to reach for when you say things like \"ship this\" or \"why is this broken\". Without it you invoke each skill by name yourself.\nStakes if we pick wrong: Mostly convenience. Without routing, skills only run when you type them; with it, Claude may auto-invoke a skill you did not want.\nRecommendation: A because this is a one-time, reversible append and it makes the rest of gstack discoverable from plain-English requests.\nNote: options differ in kind, not coverage — no completeness score.\nPros / cons:\nA) Add routing rules (recommended)\n ✅ Plain-English requests like \"review this diff\" route to the right skill without memorizing names\n ✅ One short section appended to CLAUDE.md, trivially removable later\n ❌ Adds a committed CLAUDE.md change to the repo; in plan mode this is queued until the plan is approved\nB) No thanks, invoke manually\n ✅ No repo changes; CLAUDE.md stays exactly as the fixture defines it\n ✅ Full control over when a skill runs\n ❌ You must remember and type each /skill name yourself\nNet: Convenience and discoverability versus a small committed change to CLAUDE.md.", + "header": "Routing", + "multiSelect": false, + "options": [ + { + "label": "Add routing rules (recommended)", + "description": "Append the gstack skill-routing section to CLAUDE.md and commit it. ✅ Plain-English requests route to the right skill automatically. ✅ Small, reversible append. ❌ Adds a committed CLAUDE.md change (deferred while plan mode is active)." + }, + { + "label": "No thanks, invoke manually", + "description": "Record routing_declined=true in gstack config; re-enable later via gstack-config. ✅ No repo changes. ✅ Full manual control over skill invocation. ❌ You type each /skill name yourself." + } + ] + } + ], + "answered": true, + "failed": false, + "answers": { + "gstack works best when your project's CLAUDE.md includes skill routing rules. Add them?\n\nD1 — Add gstack skill routing rules to CLAUDE.md?\nProject/branch/task: gstack-plan-count on main, starting /plan-ceo-review of PLAN.md (saved project views).\nELI10: gstack ships a dozen skills (review, ship, investigate, etc.). A short routing table in CLAUDE.md tells Claude which skill to reach for when you say things like \"ship this\" or \"why is this broken\". Without it you invoke each skill by name yourself.\nStakes if we pick wrong: Mostly convenience. Without routing, skills only run when you type them; with it, Claude may auto-invoke a skill you did not want.\nRecommendation: A because this is a one-time, reversible append and it makes the rest of gstack discoverable from plain-English requests.\nNote: options differ in kind, not coverage — no completeness score.\nPros / cons:\nA) Add routing rules (recommended)\n ✅ Plain-English requests like \"review this diff\" route to the right skill without memorizing names\n ✅ One short section appended to CLAUDE.md, trivially removable later\n ❌ Adds a committed CLAUDE.md change to the repo; in plan mode this is queued until the plan is approved\nB) No thanks, invoke manually\n ✅ No repo changes; CLAUDE.md stays exactly as the fixture defines it\n ✅ Full control over when a skill runs\n ❌ You must remember and type each /skill name yourself\nNet: Convenience and discoverability versus a small committed change to CLAUDE.md.": "Add routing rules (recommended)" + }, + "unansweredQuestionIndices": [], + "answeredAt": "2026-09-30T12:15:11.359Z" + }, + { + "sessionId": "2db5ee57-7fe2-4d95-8b3f-8ad9334707f2", + "toolUseId": "toolu_018Q8WiyPK8HAoyU89dWYwp8", + "questions": [ + { + "question": "D2 — Which review mode for the saved project views plan?\nProject/branch/task: gstack-plan-count on main, reviewing PLAN.md (saved project views).\nELI10: The mode sets my posture for the rest of the review. Expansion means I push for the bigger product. Selective means I harden what you wrote and offer add-ons one at a time for you to accept or reject. Hold means I only stress-test what is there. Reduction means I look for what to cut.\nStakes if we pick wrong: Too big and the review buries a 2-week feature under a quarter of platform work; too small and we ship personal-only views, then migrate the schema later when the team asks for shared ones.\nRecommendation: SELECTIVE EXPANSION because the plan is an added capability at ~12-14 files, and its one real gap (goal says team-wide pain, schema says per-member) is exactly the kind of cherry-pick this mode is built for: hold the CRUD core, offer shared views and deep links as opt-ins, reject the rest.\nNote: options differ in kind, not coverage — no completeness score.\nPros / cons:\nA) SELECTIVE EXPANSION (recommended)\n ✅ Keeps table + CRUD + picker fixed while you decide each add-on (shared views, default view, share links) individually\n ✅ Still runs the full HOLD rigor: error map, stale-filter failure modes, tests, observability\n ❌ More questions than HOLD; each add-on is a separate accept/defer/skip decision\nB) SCOPE EXPANSION\n ✅ Designs views as first-class objects from day one: personal, shared, defaults, automation targets\n ✅ Avoids a later schema migration if shared views are inevitable\n ❌ Likely pushes past 15 files and past the two-week pilot the plan budgets for\nC) HOLD SCOPE\n ✅ Fastest path to the pilot; maximum rigor on exactly the four bullets you wrote\n ✅ Cleanest read on whether members reuse views at all before adding surface area\n ❌ Locks in per-member-only schema; shared views become a migration if the pilot asks for them\nD) SCOPE REDUCTION\n ✅ Would strip to the smallest testable core (save + reopen, no update endpoint)\n ✅ Useful if the pilot is purely a demand test\n ❌ The plan is already small; cutting further removes rename/edit that users expect on day one\nNet: How much of the shared-view question we settle now versus after the pilot.", + "header": "Review mode", + "multiSelect": false, + "options": [ + { + "label": "SELECTIVE EXPANSION (recommended)", + "description": "Hold the submitted core; offer each expansion (shared views, default view, share links) as a separate add/defer/skip. ✅ Core stays fixed. ✅ Full HOLD rigor still applies. ❌ More decisions than HOLD." + }, + { + "label": "SCOPE EXPANSION", + "description": "Rethink views as first-class objects; recommend expansions enthusiastically. ✅ No later schema migration. ✅ Closer to the 12-month ideal now. ❌ Likely exceeds 15 files and the two-week pilot." + }, + { + "label": "HOLD SCOPE", + "description": "Preserve the four bullets exactly; maximum rigor on failures, edge cases, tests, observability. ✅ Fastest to pilot. ✅ Clean demand signal. ❌ Per-member-only schema locked in." + }, + { + "label": "SCOPE REDUCTION", + "description": "Propose the minimum viable core and cut only with your approval. ✅ Smallest testable slice. ✅ Pure demand test. ❌ Plan is already small; cuts remove expected edit/rename." + } + ] + }, + { + "question": "D3 — Enable cross-project learnings search?\nProject/branch/task: gstack-plan-count on main; one-time gstack config, not a plan decision.\nELI10: gstack keeps notes from past sessions (quirks, pitfalls, patterns). With this on, reviews on this machine can also read notes written while working on your other projects. Everything stays local.\nStakes if we pick wrong: Off means slower compounding across projects. On means notes from one client's codebase could surface while working on another's.\nRecommendation: Enable because this looks like a solo setup and local-only recall is low risk.\nNote: options differ in kind, not coverage — no completeness score.\nPros / cons:\nA) Enable cross-project learnings (recommended)\n ✅ Patterns learned in one repo carry into reviews on the next one automatically\n ✅ Stays on this machine; nothing is uploaded or shared\n ❌ Not ideal if you juggle multiple client codebases that must stay separate\nB) Keep learnings project-scoped\n ✅ Hard wall between projects; no cross-contamination of notes\n ✅ Simplest mental model for multi-client work\n ❌ Each project relearns the same lessons from scratch\nNet: Compounding across projects versus strict per-project isolation.", + "header": "Learnings", + "multiSelect": false, + "options": [ + { + "label": "Enable cross-project learnings (recommended)", + "description": "Set cross_project_learnings=true. ✅ Lessons carry across your repos. ✅ Local only. ❌ Less suitable for separate client codebases." + }, + { + "label": "Keep learnings project-scoped", + "description": "Set cross_project_learnings=false. ✅ Strict per-project isolation. ✅ Simple for multi-client work. ❌ Each project relearns from zero." + } + ] + } + ], + "answered": false, + "failed": false + } + ], + "assistantMessages": [] + } +} diff --git a/test/fixtures/ceo-mode-clipped-review-local.json b/test/fixtures/ceo-mode-clipped-review-local.json new file mode 100644 index 000000000..48528527c --- /dev/null +++ b/test/fixtures/ceo-mode-clipped-review-local.json @@ -0,0 +1,71 @@ +{ + "source": "Local paid diagnostic run 2026-09-30 (bun test test/skill-e2e-plan-ceo-mode-routing.test.ts -t \"HOLD SCOPE\", Claude Code 2.1.251): one native call bundled routing, learnings and mode; its Submit review was taller than the viewport, so the heading and first question never rendered and the mode question displayed truncated with an ellipsis. The harness never submitted HOLD SCOPE and the case failed on its posture budget.", + "screen": " │ B) Keep learnings project-scoped\n │ ✅ Hard isolation between codebases; nothing from another repo ever appears here\n │ ✅ Safest default when you work across multiple clients or employers\n │ ❌ Each new project starts cold and relearns the same environment quirks\n │ Net: faster compounding vs strict per-repo isolation.\n → Enable cross-project (recommended)\n │ ● D3 — MODE: Which review mode for the saved-views plan?\n │ Project/branch/task: gstack-plan-count-oRKiaK on main; plan adds a saved_views table, CRUD endpoints, and a picker\n │ beside task filters.\n │ ELI10: The plan is an added capability on an existing product, roughly 12 changed files (estimate; no code in this\n │ checkout). It is right-shaped but leaves three edges undefined: what the list opens on (last view vs default), what\n │ happens when a saved filter references a deleted assignee or label, and whether views are personal-only forever or\n │ the schema should leave room for team-shared views. The mode decides how hard I push on scope: expand,\n │ cherry-pick, hold, or cut.\n │ Stakes if we pick wrong: Expand too far and a two-week pilot feature becomes a quarter of work; hold too tight and\n │ the schema ships without room for sharing, forcing a migration later.\n │ Recommendation: SELECTIVE EXPANSION because those three edges are cheapest to decide while the migration is being\n │ written, and cherry-picking lets you accept or decline each one on its own without inflating the pilot.\n │ Note: options differ in kind, not coverage — no completeness score.\n │ Pros / cons:\n │ A) SELECTIVE EXPANSION (recommended)\n │ ✅ Hardens the current scope AND offers each expansion (open-on-last-view, sharing-ready schema, deleted-ref\n │ handling) as a separate yes/no\n │ ✅ Keeps the pilot small while avoiding a schema you regret in six months\n │ ❌ More decision prompts than HOLD SCOPE; you answer one question per candidate\n │ B) SCOPE EXPANSION\n │ ✅ Aims at the full ideal: team-shared views, default view per project, deep links, notifications on views\n │ ✅ Best if saved views are meant to become a platform primitive, not a pilot\n │ ❌ Turns a two-week pilot into a multi-sprint feature before you know anyone reuses views\n │ C) HOLD SCOPE\n │ ✅ Maximum rigor on exactly what is written: error paths, tests, observability, no additions\n │ ✅ Fastest path to an implementable spec for the pilot as scoped\n │ ❌ The three undefined edges get flagged as …\n → HOLD SCOPE\n\nReady to submit your answers?\n\n❯ 1. Submit answers\n 2. Cancel\n", + "transcript": { + "status": "ready", + "calls": [ + { + "sessionId": "a56b9730-c001-4931-a6f0-d33cb55eac7b", + "toolUseId": "toolu_01AxxnwBhHQyNYqtP3rzwZ4s", + "questions": [ + { + "question": "D1 — ADMIN-1: Add gstack skill routing rules to this project's CLAUDE.md?\nProject/branch/task: gstack-plan-count-oRKiaK on main, reviewing the saved-views plan.\nELI10: gstack skills work best when CLAUDE.md tells Claude which skill to reach for (bugs → /investigate, strategy → /plan-ceo-review, etc). This is a one-time onboarding prompt for this project. Note: we are in plan mode, so if you pick A the CLAUDE.md append and commit happen after plan mode exits, not now.\nStakes if we pick wrong: Without routing, you invoke each skill by name manually; with it, a fixture repo gets an extra committed section it may not want.\nRecommendation: A because routing makes skill use automatic and the section is small and reversible.\nNote: options differ in kind, not coverage — no completeness score.\nPros / cons:\nA) Add routing rules (recommended)\n ✅ Future requests auto-route to the right gstack skill without remembering names\n ✅ One small appended section, plain markdown, easy to delete later\n ❌ Adds a commit to a fixture repo that currently has a single seed commit\nB) No thanks, invoke manually\n ✅ Keeps CLAUDE.md exactly as the fixture author wrote it\n ✅ Zero writes to the repo; only a local ~/.gstack config flag is set\n ❌ You must remember and type skill names yourself every time\nNet: convenience for future sessions vs keeping the fixture repo untouched.", + "header": "Routing", + "multiSelect": false, + "options": [ + { + "label": "Add routing rules (recommended)", + "description": "✅ Auto-routes future requests to the matching gstack skill. ✅ Small reversible markdown section. ❌ Adds a commit to the fixture repo (deferred until plan mode exits)." + }, + { + "label": "No thanks, invoke manually", + "description": "✅ CLAUDE.md stays exactly as the fixture author wrote it. ✅ Only a local config flag is written. ❌ You type skill names yourself every time." + } + ] + }, + { + "question": "D2 — ADMIN-2: Let gstack search learnings from your other local projects?\nProject/branch/task: gstack-plan-count-oRKiaK on main, reviewing the saved-views plan.\nELI10: gstack logs small lessons per project (command quirks, pitfalls). Cross-project mode lets this review also read lessons logged in your other repos on this machine. Nothing leaves the machine. Good for solo devs; skip if you juggle client codebases that must stay separate.\nStakes if we pick wrong: Enabled on a multi-client machine could surface a client-specific pattern in the wrong repo; disabled on a solo machine just means fewer prior-learning hits.\nRecommendation: A because this is a local machine flag and you appear to be one developer across these repos.\nNote: options differ in kind, not coverage — no completeness score.\nPros / cons:\nA) Enable cross-project learnings (recommended)\n ✅ Reviews get smarter faster by reusing lessons from every repo on this machine\n ✅ Stays fully local; a single config flag you can flip back any time\n ❌ A learning from an unrelated codebase could be surfaced where it does not apply\nB) Keep learnings project-scoped\n ✅ Hard isolation between codebases; nothing from another repo ever appears here\n ✅ Safest default when you work across multiple clients or employers\n ❌ Each new project starts cold and relearns the same environment quirks\nNet: faster compounding vs strict per-repo isolation.", + "header": "Learnings", + "multiSelect": false, + "options": [ + { + "label": "Enable cross-project (recommended)", + "description": "✅ Reuses lessons from all local repos so reviews compound faster. ✅ Local-only flag, reversible. ❌ Unrelated-repo lesson could surface here." + }, + { + "label": "Keep project-scoped", + "description": "✅ Hard isolation between codebases. ✅ Safest for multi-client work. ❌ Every project starts cold." + } + ] + }, + { + "question": "D3 — MODE: Which review mode for the saved-views plan?\nProject/branch/task: gstack-plan-count-oRKiaK on main; plan adds a saved_views table, CRUD endpoints, and a picker beside task filters.\nELI10: The plan is an added capability on an existing product, roughly 12 changed files (estimate; no code in this checkout). It is right-shaped but leaves three edges undefined: what the list opens on (last view vs default), what happens when a saved filter references a deleted assignee or label, and whether views are personal-only forever or the schema should leave room for team-shared views. The mode decides how hard I push on scope: expand, cherry-pick, hold, or cut.\nStakes if we pick wrong: Expand too far and a two-week pilot feature becomes a quarter of work; hold too tight and the schema ships without room for sharing, forcing a migration later.\nRecommendation: SELECTIVE EXPANSION because those three edges are cheapest to decide while the migration is being written, and cherry-picking lets you accept or decline each one on its own without inflating the pilot.\nNote: options differ in kind, not coverage — no completeness score.\nPros / cons:\nA) SELECTIVE EXPANSION (recommended)\n ✅ Hardens the current scope AND offers each expansion (open-on-last-view, sharing-ready schema, deleted-ref handling) as a separate yes/no\n ✅ Keeps the pilot small while avoiding a schema you regret in six months\n ❌ More decision prompts than HOLD SCOPE; you answer one question per candidate\nB) SCOPE EXPANSION\n ✅ Aims at the full ideal: team-shared views, default view per project, deep links, notifications on views\n ✅ Best if saved views are meant to become a platform primitive, not a pilot\n ❌ Turns a two-week pilot into a multi-sprint feature before you know anyone reuses views\nC) HOLD SCOPE\n ✅ Maximum rigor on exactly what is written: error paths, tests, observability, no additions\n ✅ Fastest path to an implementable spec for the pilot as scoped\n ❌ The three undefined edges get flagged as risks but not offered as additions; schema may need a later migration for sharing\nD) SCOPE REDUCTION\n ✅ Strips to the minimum (e.g. remember last filter, no named views) to test the premise cheapest\n ✅ Lowest cost if you doubt members will bother naming views at all\n ❌ Loses the multi-context use case (switching between named lists) that the goal explicitly names\nNet: how much of the six-month shape you want to settle now versus after the pilot proves reuse.", + "header": "Review mode", + "multiSelect": false, + "options": [ + { + "label": "SELECTIVE EXPANSION (recommended)", + "description": "✅ Harden current scope, then a separate yes/no for each of the three undefined edges. ✅ Pilot stays small, schema avoids regret. ❌ More prompts than HOLD." + }, + { + "label": "SCOPE EXPANSION", + "description": "✅ Go for the full ideal: shared views, project defaults, deep links. ✅ Right if views become a platform primitive. ❌ Pilot becomes multi-sprint before reuse is proven." + }, + { + "label": "HOLD SCOPE", + "description": "✅ Max rigor on exactly what is written; fastest to implementable spec. ✅ No additions. ❌ Undefined edges flagged as risks only; sharing may need a later migration." + }, + { + "label": "SCOPE REDUCTION", + "description": "✅ Strip to remember-last-filter to test the premise cheapest. ✅ Lowest cost if naming views is doubtful. ❌ Drops the multi-context case the goal names." + } + ] + } + ], + "answered": false, + "failed": false + } + ], + "assistantMessages": [] + } +} diff --git a/test/fixtures/design-consultation-section-design-md.md b/test/fixtures/design-consultation-section-design-md.md new file mode 100644 index 000000000..2cb5b982e --- /dev/null +++ b/test/fixtures/design-consultation-section-design-md.md @@ -0,0 +1,279 @@ +--- +# gstack: design-md-format=spec +name: Ops Analytics Dashboard (working name) +description: Warm-gray paper ground, ink type, hairline rules, one signal colour reserved for state. A dispatch board, not a card deck. +colors: + primary: "#1A1C1A" # ink; primary buttons, strong rules, pinned sum lines + on-primary: "#F4F4F1" + surface: "#F4F4F1" # page ground; warm gray, near-zero chroma (not cream) + surface-raised: "#FFFFFF" # table sheets, panels, popovers + surface-sunken: "#EBEBE7" # table header row, zebra rows, disabled fields + border: "#D7D7D1" # hairline rules between rows and panels (decorative only) + text: "#1A1C1A" + text-muted: "#5E625E" # 5.6:1 on surface; also the input border colour (3:1 rule) + accent: "#1F4E79" # marine blue; links, focus ring, selected row, unfold connector + success: "#2E6B3F" # quiet on purpose + warning: "#8A5F00" # dried mustard; 5.1:1 on surface + error: "#C42B2B" # the one loud colour on the page + dark-primary: "#E9E9E4" + dark-on-primary: "#161715" + dark-surface: "#161715" + dark-surface-raised: "#1E1F1D" + dark-surface-sunken: "#101110" + dark-border: "#2E302D" + dark-text: "#E9E9E4" + dark-text-muted: "#9C9E98" + dark-accent: "#7FB2E5" + dark-success: "#6FBF87" + dark-warning: "#E0A93B" + dark-error: "#FF6B5A" +typography: + display: + fontWeight: 600 + fontSize: 1.5rem + lineHeight: 1.2 + letterSpacing: -0.01em + kpi: + fontWeight: 600 + fontSize: 2.25rem + lineHeight: 1.05 + letterSpacing: -0.02em + fontFeature: tnum, zero + body: + fontWeight: 400 + fontSize: 0.875rem + lineHeight: 1.5 + table: + fontWeight: 400 + fontSize: 0.8125rem + lineHeight: 1.25 + fontFeature: tnum + label: + fontWeight: 600 + fontSize: 0.6875rem + lineHeight: 1.2 + letterSpacing: 0.04em + textTransform: uppercase + mono: + fontWeight: 400 + fontSize: 0.8125rem + lineHeight: 1.4 + fontFeature: tnum, zero +rounded: + sm: 2px + md: 4px + lg: 6px + full: 9999px +spacing: + xs: 4px + sm: 8px + md: 12px + lg: 16px + xl: 24px + 2xl: 32px + 3xl: 48px +components: + button-primary: + backgroundColor: "{colors.primary}" + textColor: "{colors.on-primary}" + rounded: "{rounded.md}" + height: 32px + paddingX: "{spacing.md}" + button-primary-hover: + backgroundColor: "#2E312E" + button-secondary: + backgroundColor: "{colors.surface-raised}" + textColor: "{colors.text}" + borderColor: "{colors.text-muted}" + rounded: "{rounded.md}" + height: 32px + button-danger: + backgroundColor: "{colors.error}" + textColor: "{colors.surface-raised}" + rounded: "{rounded.md}" + height: 32px + input: + backgroundColor: "{colors.surface-raised}" + borderColor: "{colors.text-muted}" + textColor: "{colors.text}" + rounded: "{rounded.sm}" + height: 32px + paddingX: "{spacing.sm}" + focus-ring: + outlineColor: "{colors.accent}" + outlineWidth: 2px + outlineOffset: 2px + panel: + backgroundColor: "{colors.surface-raised}" + borderColor: "{colors.border}" + rounded: "{rounded.md}" + padding: "{spacing.lg}" + table-header: + backgroundColor: "{colors.surface-sunken}" + textColor: "{colors.text-muted}" + height: 32px + table-row: + height: 32px + borderColor: "{colors.border}" + table-row-compact: + height: 28px + table-row-wall: + height: 44px + nav-link: + textColor: "{colors.text-muted}" + height: 32px + nav-link-active: + textColor: "{colors.text}" + backgroundColor: "{colors.surface-sunken}" + section-rule: + borderColor: "{colors.primary}" + borderWidth: 2px + status-dot: + size: 8px + rounded: "{rounded.full}" +--- + +# Ops Analytics Dashboard (working name) + +## Overview + +**Creative North Star:** Industrial/Utilitarian in a dispatch-board register. Every pixel of chroma is information, so an ops lead sees what is off-target, and by how much, before finishing the first read. +**Product context:** B2B analytics dashboard for operations teams (ops managers, analysts, shift leads, on-call staff) in logistics, support, fulfilment, field service and platform ops. They watch throughput, queue depth, SLA attainment, incidents and staffing against targets and drill from a headline number into the rows behind it. Web app dashboard, greenfield, no prior brand. +**Mode per surface:** +- Operate: the dashboard, tables, filters, alerts. The primary surface; everything below is tuned for it. +- Read: scheduled reports and incident write-ups. Same tokens, 72ch measure, body at 1rem. +- Persuade: a small marketing site later. Same palette and faces, more whitespace, display up to 3rem, still left-aligned. +- Experience: none. +**Reference sites:** none. Competitive research was declined; this system comes from the users' own world (rail timetables, shift boards, dispatch sheets, control-room mimic boards), not from remembered competitor screens. +**The one thing to remember (working answer, agent-selected, confirm with the team):** "Nothing on this screen is decoration. You knew what was wrong, and how wrong, before you finished reading." +**Key characteristics:** +- A warm gray sheet with ink type and hairline rules. It reads as a document the team owns, not a product they rent. +- The headline band is a row of numbers with labels underneath, not a row of tiles. +- Red appears rarely, so when it appears you look at it. +- Dense by default. Row height 32px, 13px table type, tabular figures right-aligned to the decimal. +- No drop shadows on the sheet. Depth exists only on overlays. + +## Colors + +**Strategy:** Restrained. Neutrals plus one interactive hue (marine blue) plus three status hues. Nothing else. +**Light or dark:** Light is the default because the primary use scene is an eight-hour desk session under office lighting, where a dark UI produces glare halos and forces re-adaptation every time the eye leaves the screen. Dark is a real mode, not an inversion, auto-selected for the Wall preset (ops-room screens viewed from distance) and offered on phones for night on-call. + +Neutrals derive from a near-zero-chroma warm gray, not blue-gray. `surface` `#F4F4F1` is the ground; `surface-raised` white is where data sits; `surface-sunken` marks table headers, zebra rows and disabled fields. The ground is deliberately not cream: a yellow ground shifts the perceived hue of `warning` toward `error`. + +`primary` is ink. Primary buttons, section rules and pinned sum lines are near-black, which keeps all saturated colour free for meaning. `accent` marine blue signals interaction only: links, the focus ring, the selected row, the connector rule on an unfolded drill-down. Blue is the one hue with no status meaning, which is why it and only it may signal "you can act here". + +Status hues are ranked by loudness on purpose. `success` `#2E6B3F` is quiet (nothing to see). `warning` `#8A5F00` is a dried mustard chosen to separate from both red and green for deutan and protan viewers and to pass 5.1:1 as text on the ground. `error` `#C42B2B` is the only loud colour on the page. Status is always encoded three ways: hue, a gutter glyph (▲ ▼ ■) and a label or underline. Colour is never the only signal. + +Dark theme preserves the same hierarchy: `dark-surface-raised` sits one step lighter than `dark-surface`, `dark-surface-sunken` one step darker, and separation is still a 1px rule, never a shadow or glow. Status hues are lifted in lightness (`dark-error` `#FF6B5A`) because dark surfaces swallow saturation. Text stays warm off-white so day and night feel like one instrument under two lights. + +Contrast (computed against the ground): text 16:1, text-muted 5.6:1, accent 7.7:1, success 5.8:1, warning 5.1:1, error 5.1:1. Dark: text 14.8:1, muted 6.7:1, accent 8.1:1, success 8.1:1, warning 8.5:1, error 6.4:1. `border` `#D7D7D1` is 1.3:1 and is therefore reserved for decorative hairlines; input and control boundaries use `text-muted`. + +## Typography + +**Source world and register:** timetables, dispatch sheets, departure boards, instrument panels. Mode: Operate. The register is a plain, sturdy grotesk for words and a tabular face for numbers. No serif: on this ground with a red status colour a serif display is the cream/serif/terracotta default, and serif hairlines vanish on a wall screen at four metres. + +**Font selection: PENDING VERIFICATION.** No font listing could be checked in this session (no web search, no shell). Per the consultation's font-verification rule the `fontFamily` values are omitted from the front matter above and no loading URL is given. Verify each candidate's exact name, weights, license and loading URL on its official Google Fonts or Fontshare listing before adopting; if a candidate fails, take the named alternate. Do not substitute `system-ui`, Inter, Roboto or Arial as the design intent in the meantime; a generic `sans-serif` / `monospace` stack in development is acceptable only until verification lands. + +| Role | Candidate (pending) | Alternate (pending) | Weights | Used for | +|---|---|---|---|---| +| Display | Cabinet Grotesk (Fontshare) | General Sans (Fontshare) | 500, 700 | Page titles at 1.5rem, section titles at 1.25rem, marketing headlines up to 3rem. Never body copy. | +| Body and UI | Source Sans 3 (Google Fonts) | IBM Plex Sans (Google Fonts; on the overused list, permitted here as body/UI on an Operate surface because it was drawn for dense data screens and has tabular figures) | 400, 600 | Table cells at 0.8125rem, controls and copy at 0.875rem, Read surfaces at 1rem. Requires `tnum`. | +| Label | same face as Body | same | 600 | Column headers, KPI labels, metadata: 0.6875rem, uppercase, 0.04em tracking. | +| Mono | JetBrains Mono (Google Fonts) | IBM Plex Mono (Google Fonts) | 400, 600 | IDs, SKUs, ISO timestamps, log lines, shift notes, and the hero KPI numerals (see Risks). Requires `tnum` and `zero`. | + +**Scale:** 11 / 13 / 14 / 16 / 20 / 24 / 36 px. Each level differs by size, not just weight. KPI numerals are 2.25rem at 600 with -0.02em tracking; the Wall preset scales them to 4.5rem and body to 1.125rem. Body never drops below 12px on desktop or 14px on phone. + +**Numerals:** every numeric column and every KPI uses `font-variant-numeric: tabular-nums slashed-zero`, right-aligned, decimal-aligned. A column of numbers must read as a shape. + +**Loading strategy (once verified):** self-host WOFF2 with `font-display: swap`, preload the body face only, subset to Latin. Two families and one mono, no more. + +## Layout + +**Grid-disciplined.** The dashboard is fluid; width is data. + +- Desktop (≥1280px): 12-column fluid grid, 16px gutters, 24px page margins. Left rail 240px, collapsible to 56px (icons plus tooltips). No max width on Operate surfaces. +- Laptop (1024 to 1279px): 8 columns, rail collapsed by default. +- Tablet and phone (<1024px): single column. Rail becomes a top bar and drawer. KPI band wraps 2-up. Tables scroll horizontally with the first column and header pinned. +- Wall preset (≥1920px, kiosk): nav hidden, 1.25× type scale, row height 44px, dark theme by default, no hover states. +- Read surfaces: 72ch measure, centred column, left-aligned text. +- Marketing: 1200px max width, same grid, no centred headings. + +**Rhythm:** 4px base, 8px step. Panel padding 16px, grid gap 16px, section gap 32px, page section gap 48px. Row height 32px default, 28px compact, 44px wall, selectable from a three-position density control in the toolbar (Compact / Standard / Wall). Interactive elements outside tables keep a 32px minimum height and 40px on touch. + +**The headline band (adopted from the independent voice):** the top of every dashboard page is one typographic row of six to eight hero figures, each with its label beneath in the label style and its target set in `text-muted` to the right (`4,812 / 5,000`). Deviation is shown by the figure itself changing colour, a gutter glyph and an underline. No tiles, no icons, no sparklines. + +**Drill-down as unfold (adopted from the independent voice):** clicking a hero figure unfolds the rows behind it directly beneath the band. The figure stays pinned as a sum line, a 1px `accent` rule connects the two, and each further level pins another sum line. Breadcrumbs are the stack of pinned sum lines. The user never loses the number they were looking at. + +**Intentional grid break:** exactly one. Section titles sit on top of a 2px ink rule that runs the full content width, breaking the column gutter the way a ledger heading sits on its column line. + +## Elevation & Depth + +The sheet is flat. Panels, tables and the headline band are separated by 1px `border` rules and by `surface-sunken` tints, never by shadows. Depth exists only where something genuinely floats: + +- Popover, menu, tooltip: `0 4px 12px rgba(26, 28, 26, 0.12)` plus a 1px `border`. +- Dialog, drawer: `0 12px 32px rgba(26, 28, 26, 0.18)` plus a 1px `border`, over a `rgba(26, 28, 26, 0.32)` scrim. +- Dark theme: shadows drop to `rgba(0, 0, 0, 0.5)` at the same offsets; the 1px `dark-border` does the work. + +No zero-offset glow, no coloured halo, no inset highlight, no frosted glass. + +## Shapes + +Small radii throughout so nothing reads as a bubble. + +- `sm` 2px: inputs, selects, tags, table cells with a tint. +- `md` 4px: buttons, panels, popovers. +- `lg` 6px: dialogs and drawers. +- `full`: status dots and avatar marks only. Never on buttons. +- Nested element radius = outer radius minus the gap. A 2px-radius tag inside a 4px panel with 2px inset is correct; a 4px tag inside a 4px panel is not. + +## Components + +Every component ships all states: default, hover, focus-visible, active, disabled, loading, empty, error, and long-content. States below are the invariants; the Wall preset removes hover states and scales heights. + +- **Button primary:** ink on ground, 32px, 12px horizontal padding, 4px radius, 600 weight at 0.8125rem. Hover `#2E312E`. Active darkens to `#0F100F`. Focus-visible: 2px `accent` outline, 2px offset. Disabled: `surface-sunken` background, `text-muted` text, no border. One primary per view. +- **Button secondary:** white, 1px `text-muted` border, ink text. Hover `surface-sunken`. +- **Button ghost:** no border, `accent` text. Hover underlines. Used for inline row actions. +- **Button danger:** `error` background, white text. Only on the confirming step of a destructive action, never in a toolbar. +- **Input / select:** white, 1px `text-muted` border, 2px radius, 32px, 8px padding. Focus: border becomes `accent` plus the focus ring. Error: border `error`, message below in `error` at label size, with an icon. Disabled: `surface-sunken`, no border. Labels above the field in label style; help text below in `text-muted`. +- **Table:** sticky header on `surface-sunken` in label style; 32px rows separated by 1px `border`; numeric columns right-aligned with `tnum`; text columns left-aligned; first column pinned on horizontal scroll. Hover row `surface-sunken`; selected row `accent` at 8% tint with a 2px `accent` left rule inside the cell padding (the rule is inside a rectangular row, not on a rounded card). Sort indicator is a glyph, not a colour. Loading: skeleton rows in `surface-sunken`, no shimmer. Empty: one sentence saying what would be here and the action that fills it. Error: the failing panel keeps its frame and shows the message inline with a retry. +- **KPI figure:** mono face, 2.25rem, 600, `tnum zero`; label beneath; target to the right in `text-muted`; deviation colours the figure, adds a gutter glyph (▲ over, ▼ under, ■ on target) and a 2px underline in the same hue. On target, the figure stays ink. Never boxed. +- **Status:** 8px dot plus label, or figure recolour plus glyph. Success is quiet, warning is mustard, error is red. Never colour alone. +- **Side nav link:** 32px, `text-muted`, 0.8125rem. Hover ink text. Active: ink text on `surface-sunken`, no accent bar. Collapsed rail shows icons at 20px with tooltips. +- **Filter bar:** a single 40px row of inputs and chips under the page title; applied filters render as removable 2px-radius chips in `surface-sunken`. Never a modal. +- **Toast:** bottom-left, white, 1px `border`, 4px radius, status glyph, auto-dismiss 6s except errors, which persist. +- **Section rule:** 2px ink rule with the section title sitting on it, left-aligned. + +## Do's and Don'ts + +- Do: set every numeric column with `tabular-nums`, right-aligned, decimal-aligned. +- Do: encode every status three ways (hue, glyph, label or underline). +- Do: separate panels with 1px rules and tints; reserve shadows for things that float. +- Do: keep one primary button per view and keep it ink. +- Do: design empty, loading, error and long-content states before shipping a component. +- Don't: put a KPI in a tile with an icon, sparkline and delta chip. The figure is the component. +- Don't: use blue for anything that is not interactive, or any status hue for anything that is not status. +- Don't: nest a card in a card, or put a coloured left border on a rounded card. +- Don't: switch the ground to cream or the display to a serif; that is the stock "warm editorial" look and it breaks the amber/red separation. +- Don't: choose dark because it is a tool. Dark is for the wall and the night shift, decided by the use scene. +- Don't: add a kicker above a heading, an icon tile above a section, or a gradient anywhere. + +## Motion + +- **Approach:** minimal-functional. Motion exists to keep the user's eye on the number they were reading. +- **Easing:** enter(ease-out) exit(ease-in) move(ease-in-out) +- **Duration:** micro(80ms) hover and pressed states; short(160ms) menus, popovers, tooltips; medium(240ms) drawer, unfold; long(400ms) reserved, currently unused. +- **The one authored moment:** the ledger unfold. Clicking a hero figure slides the rows open beneath it over 240ms ease-out while the figure stays pinned and the `accent` connector rule draws from the figure down to the table header. When a figure crosses a threshold on live data, its colour, glyph and underline transition over 240ms, no flash, no pulse. +- `prefers-reduced-motion`: all durations drop to 0 except opacity fades at 80ms. + +## Decisions Log +| Date | Decision | Rationale | +|------|----------|-----------| +| 2026-09-29 | Initial design system created | Created by /design-consultation from product context (B2B ops analytics dashboard); competitive research declined by the user; one native independent voice consulted, Codex unavailable in this harness | +| 2026-09-29 | Light default, dark for Wall preset and night on-call | Decided by the use scene (long desk sessions under office light), not category habit | +| 2026-09-29 | Ink primary; chroma reserved for interaction and status | Serves the memorable thing: every coloured pixel is information | +| 2026-09-29 | Warm gray ground `#F4F4F1`, not cream | Cream plus red status shifts amber toward red; also avoids the cream/serif/terracotta default | +| 2026-09-29 | Headline band of figures and in-place ledger unfold | Adopted from the independent native voice; both keep the user's eye on the number | +| 2026-09-29 | Serif hero numerals rejected | Calibration look one on this palette; serif hairlines fail on wall screens at distance | +| 2026-09-29 | Fonts pending verification | No web search or shell available in-session; fontFamily omitted from tokens until Google Fonts / Fontshare listings are checked | +| 2026-09-29 | Preview deferred | Fonts unverified; the consultation's fallback defers the Phase 5 preview until faces can be verified | + diff --git a/test/helpers/auq-sdk-capture.ts b/test/helpers/auq-sdk-capture.ts index f647d97d8..5f64f4846 100644 --- a/test/helpers/auq-sdk-capture.ts +++ b/test/helpers/auq-sdk-capture.ts @@ -53,7 +53,8 @@ export function scoreAuqFormat(text: string): { present: number; total: number; * whether the ORIGINAL used the literal "because" — a soft style signal, since * the format spec prefers it and the voice rule forbids the em-dash form. * - * This does NOT touch judgeRecommendation or its pinned fixtures. + * This does NOT touch judgeRecommendation or its pinned fixtures. A judge + * failure propagates with its cause; it is never reported as substance 0. */ export async function gradeAuqRecommendation( text: string, @@ -75,12 +76,8 @@ export async function gradeAuqRecommendation( } } - try { - const r = await judgeRecommendation(graded); - return { substance: r.reason_substance, present: r.present, hadLiteralBecause, reason: r.reason_text }; - } catch { - return { substance: 0, present: !!recLine, hadLiteralBecause, reason: '' }; - } + const r = await judgeRecommendation(graded); + return { substance: r.reason_substance, present: r.present, hadLiteralBecause, reason: r.reason_text }; } /** diff --git a/test/helpers/carve-guards.ts b/test/helpers/carve-guards.ts index 914bb0743..af64fb35c 100644 --- a/test/helpers/carve-guards.ts +++ b/test/helpers/carve-guards.ts @@ -387,7 +387,7 @@ do not launch the downstream skill or open a browser.`, expectedSections: ['proposal-and-preview.md'], requiredReads: ['proposal-and-preview.md'], scenario: - 'The user gave product context (a B2B analytics dashboard for ops teams) and declined the research phase. Skip browser/design tool setup. Proceed to build the complete design-system proposal, then write DESIGN.md. Produce the proposal and the DESIGN.md content.', + 'The user gave product context (a B2B analytics dashboard for ops teams), declined the research phase and declined the optional outside design voices. Skip browser/design tool setup. Proceed to build the complete design-system proposal, then write DESIGN.md and its CLAUDE.md guidance.', staticInvariants: { mustStayInSkeleton: ['## Phase 0: Pre-checks', '## Phase 1: Product Context', '## Phase 2: Research'], mustMoveToSection: ['## Phase 3: The Complete Proposal', '## Phase 6: Write DESIGN.md'], diff --git a/test/helpers/carve-section-case.ts b/test/helpers/carve-section-case.ts index f088e6f30..20263b374 100644 --- a/test/helpers/carve-section-case.ts +++ b/test/helpers/carve-section-case.ts @@ -108,11 +108,15 @@ export function registerCarveSectionCase(skill: string): void { ? '- Proceed directly with the requested engineering review; skip the optional /office-hours prerequisite. You represent the plan author, whose scope and proposed steps are in PLAN.md. At each decision, choose the complete alternative that preserves those requirements and existing contracts; choose the recommended option only among alternatives within that scope. Do not authorize optional scope, extra public input guarantees, arbitrary size limits, or optional proof projects. Decline work explicitly listed out of scope, including creating TODOs for it. Record the decision and its actual authority as the skill requires, then continue without asking a human. A demonstrated incompatibility or missing required proof still requires resolution; do not hide it or claim approval when no offered alternative meets these constraints.' : undefined, // Both plan reviews persist their required report in the reviewed plan. - reportFile: ['plan-devex-review', 'plan-eng-review'].includes(guard.skill) ? 'PLAN.md' : undefined, + // design-consultation's final output is DESIGN.md itself; a second + // REPORT.md only duplicated the proposal (census 36641820398 timeout). + reportFile: ['plan-devex-review', 'plan-eng-review'].includes(guard.skill) ? 'PLAN.md' + : guard.skill === 'design-consultation' ? 'DESIGN.md' : undefined, // This scenario produces an HTML implementation, whose complete // document need not contain any of the prose report keywords. reportMarker: guard.skill === 'design-html' ? /\s*]*>[\s\S]*?]*>[\s\S]*?<\/head\s*>[\s\S]*?]*>[\s\S]*?<\/body\s*>\s*<\/html\s*>/i + : guard.skill === 'design-consultation' ? /^# gstack: design-md-format=spec$/m : /report|review|summary|design doc|handoff/i, testName: `${guard.skill} section-loading`, runId, @@ -125,7 +129,7 @@ export function registerCarveSectionCase(skill: string): void { }); // Require the HTML artifact itself; a terminal-only claim is insufficient. // captureSectionReads already requires a successful native completion. - const reportProduced = completionMarked && (guard.skill !== 'design-html' || reportWritten); + const reportProduced = completionMarked && (!['design-html', 'design-consultation'].includes(guard.skill) || reportWritten); const missing = guard.requiredReads.filter((s) => !readSections.has(s)); // Named failure output (codex #2): skill + expected + observed. diff --git a/test/helpers/ceo-mode-option.ts b/test/helpers/ceo-mode-option.ts index d15f9823f..27e170a41 100644 --- a/test/helpers/ceo-mode-option.ts +++ b/test/helpers/ceo-mode-option.ts @@ -146,6 +146,77 @@ function hasNativePostureProse(text: string, posture: RegExp): boolean { return hasPostAnswerCeoPosture(`● ${prose}`, posture); } +const CLIPPED_PREFIX_MIN = 120; + +/** + * A review taller than the viewport can clip its heading and earlier questions + * before they ever render, and it truncates a long question with "…". Authenticate + * the visible tail from the Submit prompt backwards: every visible answer is an + * offered option, each question below the clip matches its native text (or a long + * native prefix before "…"), the mode question's target answer is visible, and only + * the topmost segment may be cut off above the viewport; a cut mode question must + * still show a long native tail. The native answer is verified again after Submit. + */ +function clippedReviewMatches(visible: string, selected: NativePlanQuestionCall, + modeQuestion: NativePlanQuestionCall['questions'][number], targetMode: CeoMode): boolean { + const compact = (text: string) => text.replace(/\s+/g, ''); + let body = compact(visible.replace(/^[ \t]*[│┃] ?/gm, '').replace(/^[ \t]*[●⏺] ?/gm, '')); + if (!body.endsWith(BARLESS_SUBMIT_END) || /[←☐☒]/.test(body)) return false; + body = body.slice(0, -BARLESS_SUBMIT_END.length); + const modeIndex = selected.questions.indexOf(modeQuestion); + for (let i = selected.questions.length - 1; i >= 0; i--) { + const question = selected.questions[i]!; + const answers = (i === modeIndex + ? question.options.filter(o => modeTitle(o.label) === targetMode.replace(/\s+/g, '')) + : question.options).map(o => `→${compact(o.label)}`).filter(answer => body.endsWith(answer)); + if (answers.length !== 1) return false; + body = body.slice(0, -answers[0]!.length); + const text = compact(question.question); + let shown = 0; + if (body.endsWith(text)) shown = text.length; + else if (body.endsWith('…')) { + for (let length = text.length - 1; length >= CLIPPED_PREFIX_MIN && !shown; length--) { + if (body.slice(0, -1).endsWith(text.slice(0, length))) shown = length + 1; + } + } + if (!shown) { + if (i > modeIndex || (i === modeIndex && body.replace(/…$/, '').length < CLIPPED_PREFIX_MIN)) return false; + return body.endsWith('…') ? text.includes(body.slice(0, -1)) : text.endsWith(body); + } + body = body.slice(0, -shown); + if (!body) return i <= modeIndex; + } + return 'Reviewyouranswers'.endsWith(body); +} + +/** + * A packet can bundle setup tabs after the mode tab. Once the mode tab is + * answered, answer each later non-mode tab of the same unsubmitted call once, + * with the navigation rule (prerequisite pick, else option 1), so Submit is reachable. + */ +export function ceoModePacketTabAnswer( + visible: string, selected: NativePlanQuestionCall | undefined, transcript: PlanCountTranscript, answered: Set, +): { question: AskUserQuestionFingerprint; index: number } | null { + if (!selected || !selected.sessionId || !selected.toolUseId || transcript.status !== 'ready' || + selected.questions.length < 2 || selected.questions.length > 4 || selected.questions.some(q => q.multiSelect)) return null; + const id = `${selected.sessionId}:${selected.toolUseId}`; + const current = transcript.calls.filter(call => `${call.sessionId}:${call.toolUseId}` === id); + if (current.length !== 1 || current[0]!.answered || current[0]!.failed || + JSON.stringify(current[0]!.questions) !== JSON.stringify(selected.questions)) return null; + const bar = posturePacketBar(visible); + if (!bar || JSON.stringify(bar.headers) !== JSON.stringify(selected.questions.map(q => q.header.trim().replace(/\s+/g, ' ')))) return null; + const modeIndex = selected.questions.findIndex(q => q.options.filter(o => modeTitle(o.label)).length >= 2); + if (modeIndex < 0 || !bar.answered[modeIndex]) return null; + const question = capturePlanCountQuestion(visible, new Set(), 0, true, selected); + const tab = question?.nativeQuestionIndex; + if (!question || question.nativeCall !== selected || tab === undefined || tab <= modeIndex || bar.answered[tab] || + JSON.stringify(question.options.map(o => o.label)) !== JSON.stringify(selected.questions[tab]!.options.map(o => o.label))) return null; + const key = `${id}:${tab}`; + if (answered.has(key)) return null; + answered.add(key); + return { question, index: planCountPrerequisitePick(question) ?? 1 }; +} + /** Finish the selected native mode packet before waiting for its answer. */ export function ceoModeSubmissionInput( visible: string, selected: NativePlanQuestionCall | undefined, targetMode: CeoMode, @@ -178,6 +249,11 @@ export function ceoModeSubmissionInput( // focused Submit prompt; the accumulated screen text then supplies the // one complete review panel, authenticated below exactly as with a bar. const heading = screenText.lastIndexOf('Review your answers'); + if (heading < 0 && compact(screenText).endsWith(BARLESS_SUBMIT_END) && + clippedReviewMatches(visible, selected, modeQuestions[0]!, targetMode)) { + submitted.add(id); + return '\r'; + } if (heading < 0 || !compact(visible).endsWith(BARLESS_SUBMIT_END) || quotedContext.test(screenText.slice(0, heading).split('\n').slice(-3).join('\n'))) return null; review = screenText.slice(heading); diff --git a/test/helpers/llm-judge.ts b/test/helpers/llm-judge.ts index 22c24a973..0b5ef4186 100644 --- a/test/helpers/llm-judge.ts +++ b/test/helpers/llm-judge.ts @@ -397,6 +397,16 @@ ${text}`, undefined, { signal }); * Format spec: scripts/resolvers/preamble/generate-ask-user-format.ts * Recommendation: because */ +export const RECOMMENDATION_JUDGE_SCHEMA = { + type: 'object', + properties: { + reason_substance: { type: 'integer', enum: [1, 2, 3, 4, 5] }, + reasoning: { type: 'string' }, + }, + required: ['reason_substance', 'reasoning'], + additionalProperties: false, +}; + export async function judgeRecommendation(askUserText: string, signal?: AbortSignal): Promise { signal?.throwIfAborted(); // Deterministic checks. The format spec requires: @@ -472,7 +482,7 @@ Respond with ONLY valid JSON: const out = await callJudge<{ reason_substance: number; reasoning: string }>( prompt, 'claude-haiku-4-5-20251001', - { signal }, + { signal, jsonSchema: RECOMMENDATION_JUDGE_SCHEMA }, ); // Defensive clamp: rubric is 1-5. If Haiku returns out-of-range or non-numeric, diff --git a/test/llm-judge-frontier.test.ts b/test/llm-judge-frontier.test.ts index 6e0c8ce28..dad4ec05f 100644 --- a/test/llm-judge-frontier.test.ts +++ b/test/llm-judge-frontier.test.ts @@ -1,6 +1,7 @@ import { afterEach, beforeEach, describe, expect, spyOn, test } from 'bun:test'; import Anthropic from '@anthropic-ai/sdk'; -import { armJudge, callJudge, JudgeRefusalError } from './helpers/llm-judge'; +import { armJudge, callJudge, JudgeRefusalError, judgeRecommendation, RECOMMENDATION_JUDGE_SCHEMA } from './helpers/llm-judge'; +import { gradeAuqRecommendation } from './helpers/auq-sdk-capture'; describe('frontier Claude judge compatibility', () => { let originalKey: string | undefined; @@ -194,6 +195,35 @@ describe('frontier Claude judge compatibility', () => { } }); + // Census 36641820398 slice 7: Haiku's free-form reply left inner quotes unescaped. + const CENSUS_UNESCAPED_REPLY = '```json\n{"reason_substance": 4, "reasoning": "The clause is concrete and option-specific ("at 1/10 every dimension has gaps and none are obviously safe to skip") but does not explicitly compare the chosen option (A) against alternatives B or C in the because-clause itself\u2014the comparison lives in the surrounding context, not in the reason block."}\n```'; + const CENSUS_BRIEF = 'D1 \u2014 Review all 7 design dimensions?\nRecommendation: A because at 1/10 every dimension has gaps and none are obviously safe to skip.\nA) All 7 dimensions (recommended)'; + + test('recommendation judge requests structured output, so quoted evidence cannot break its JSON', async () => { + create.mockResolvedValue({ stop_reason: 'end_turn', content: [{ type: 'text', + text: JSON.stringify({ reason_substance: 4, reasoning: 'Concrete ("at 1/10 every dimension has gaps") but no comparison.' }) }] } as never); + const score = await judgeRecommendation(CENSUS_BRIEF); + expect(score.reason_substance).toBe(4); + expect(score.reason_text).toBe('at 1/10 every dimension has gaps and none are obviously safe to skip.'); + expect(create.mock.calls[0][0].output_config).toEqual({ format: { type: 'json_schema', schema: RECOMMENDATION_JUDGE_SCHEMA } }); + expect(create.mock.calls[0][0].model).toBe('claude-haiku-4-5-20251001'); + }); + + test('the captured unescaped reply is unparseable free-form text and fails without a score', async () => { + create.mockResolvedValue({ stop_reason: 'end_turn', content: [{ type: 'text', text: CENSUS_UNESCAPED_REPLY }] } as never); + await expect(callJudge('score this', 'claude-haiku-4-5-20251001')).rejects.toThrow(SyntaxError); + create.mockClear(); + create.mockResolvedValue({ stop_reason: 'end_turn', content: [{ type: 'text', text: CENSUS_UNESCAPED_REPLY }] } as never); + await expect(gradeAuqRecommendation(CENSUS_BRIEF)).rejects.toThrow(SyntaxError); + expect(create).toHaveBeenCalledTimes(1); + }); + + test('a weak structured score still fails the substance bar', async () => { + create.mockResolvedValue({ stop_reason: 'end_turn', content: [{ type: 'text', + text: JSON.stringify({ reason_substance: 1, reasoning: 'Boilerplate.' }) }] } as never); + expect((await gradeAuqRecommendation('Recommendation: A because it is better.')).substance).toBe(1); + }); + test('arm judge sends no unsupported temperature to Fable', async () => { create.mockResolvedValue({ content: [ { type: 'thinking', thinking: '', signature: 'fixture' }, diff --git a/test/skill-e2e-plan-ceo-mode-routing.test.ts b/test/skill-e2e-plan-ceo-mode-routing.test.ts index 9549b5a8b..7c84dad60 100644 --- a/test/skill-e2e-plan-ceo-mode-routing.test.ts +++ b/test/skill-e2e-plan-ceo-mode-routing.test.ts @@ -44,7 +44,7 @@ import { type AskUserQuestionFingerprint, type ClaudePtySession, } from './helpers/claude-pty-runner'; -import { ceoExpansionPacingChoice, ceoExpansionPacingReady, ceoModeSubmissionInput, hasNativePostAnswerCeoPosture, holdDeferKeepIndex, nextCeoModeNavigation, nextCeoPostureContinuation } from './helpers/ceo-mode-option'; +import { ceoExpansionPacingChoice, ceoExpansionPacingReady, ceoModePacketTabAnswer, ceoModeSubmissionInput, hasNativePostAnswerCeoPosture, holdDeferKeepIndex, nextCeoModeNavigation, nextCeoPostureContinuation } from './helpers/ceo-mode-option'; import { createPlanCountFixture } from './helpers/plan-count-fixture'; import { readPlanCountTranscript, type NativePublicToolEvent, type PlanCountTranscript } from './helpers/plan-count-transcript'; import { readPendingQuestion, pendingQuestionRecorderStatus } from './helpers/plan-count-pending-question'; @@ -231,6 +231,7 @@ describeE2E('/plan-ceo-review mode routing (gate)', () => { let pacingCalls = 0; const seenDownstream = new Set(); const submittedModePackets = new Set(); + const answeredPacketTabs = new Set(); while (Date.now() - start < budgetMs) { await Bun.sleep(2500); if (session.exited()) { @@ -262,6 +263,13 @@ describeE2E('/plan-ceo-review mode routing (gate)', () => { capture('awaiting_posture', currentInput, transcript); const modeSubmit = ceoModeSubmissionInput(currentInput, question.nativeCall, c.mode, transcript, submittedModePackets, session.visibleText()); if (modeSubmit !== null) { session.send(modeSubmit); continue; } + const packetTab = ceoModePacketTabAnswer(currentInput, question.nativeCall, transcript, answeredPacketTabs); + if (packetTab) { + const input = planCountQuestionInput(currentInput, packetTab.question, packetTab.index); + if (input.includes('\r')) await selectPtyNumberedOption(session, packetTab.index); + else session.send(input); + continue; + } const pendingQuestion = readPendingQuestion(session.pendingQuestionFile, fixture.cwd, session.hermeticConfigDir, selectionStartedAt, transcript); if (pacingChoice && !ceoExpansionPacingReady(currentInput, transcript, pacingChoice, publicTools)) continue;