From 878092dd0d12a304c3fa85e63e21cb6db813d816 Mon Sep 17 00:00:00 2001 From: garrytan Date: Wed, 30 Sep 2026 23:17:19 +0000 Subject: [PATCH] test(design): plan-mode names its read list and caps its additions and summary At 6fcb0981 plan-design-review-plan-mode timed out at 300 s (9 turns): 22 cat/sed chunk reads (~50 s), then a 28 KB plan Write (~150 s), before the read-back finished. The 9a7a7e54 pass took 240 s with a 24.6 KB Write. Read SKILL.md, review-sections.md and plan.md natively in one response, keep additions under 14,000 characters and the summary within ten lines. Budgets unchanged. --- test/plan-design-sdk-fixture.test.ts | 6 +++++- test/skill-e2e-design.test.ts | 6 +++--- 2 files changed, 8 insertions(+), 4 deletions(-) diff --git a/test/plan-design-sdk-fixture.test.ts b/test/plan-design-sdk-fixture.test.ts index b6aff6592..b9884db1f 100644 --- a/test/plan-design-sdk-fixture.test.ts +++ b/test/plan-design-sdk-fixture.test.ts @@ -89,7 +89,7 @@ async function exercise(mode: 'success' | 'max-turns' | 'first-timeout' | 'secon expect(opts.signal.aborted).toBe(false); // Bind the complete actual compact-delivery prompt, not selected snippets. expect(new Bun.CryptoHasher('sha256').update(opts.prompt).digest('hex')) - .toBe('2fa957ab9d56850a1629a845d6fe0ee5a1cb7c0843ab6555b621971d270604cb'); + .toBe('56468e8f520719957c612ced2da1f5f2f3698bc4551075a429bee4816e157cdb'); expect(opts.testName).toBe(id); expect(opts.maxTurns).toBe(15); expect(opts.timeout).toBe(CAPTURE_MS); for (const key of ['model', 'tools', 'allowedTools', 'appendSystemPrompt', 'env']) expect(opts).not.toHaveProperty(key); expect(opts.prompt).toContain('Review the plan in ./plan.md'); @@ -101,6 +101,10 @@ async function exercise(mode: 'success' | 'max-turns' | 'first-timeout' | 'secon expect(opts.prompt).toContain('Write before publishing a completed walkthrough'); expect(opts.prompt).toContain('Read plan.md back to verify the saved changes'); expect(opts.prompt).toContain('Then return a brief, concrete summary'); + expect(opts.prompt).toContain('natively Read plan-design-review/SKILL.md'); + expect(opts.prompt).toContain('plan-design-review/sections/review-sections.md (the one lazy section this review requires)'); + expect(opts.prompt).toContain('including the report, under 14,000 characters'); + expect(opts.prompt).toContain('in at most ten lines'); expect(opts.prompt).toContain('execute every required pass and lazy-section Read'); expect(opts.prompt).toContain('Retain all required report fields, design decisions, diagrams, ratings, and explanations'); expect(opts.prompt).toContain('use the canonical tables and decision IDs'); diff --git a/test/skill-e2e-design.test.ts b/test/skill-e2e-design.test.ts index b75ecd606..808a2a595 100644 --- a/test/skill-e2e-design.test.ts +++ b/test/skill-e2e-design.test.ts @@ -461,10 +461,10 @@ Build a user dashboard that shows account stats, recent activity, and settings. Review the plan in ./plan.md. Its design gaps are vague "clean, modern UI" and "cards and icons", a "hero section with gradient" (AI slop), and missing empty, error, loading, responsive, and accessibility behavior. Use this non-interactive delivery sequence: -1. Skip the preamble bash block and any AskUserQuestion calls. Read every lazy section the workflow requires. Review all 7 design passes. Rate each scored design dimension 0-10 and explain what would make it a 10; preserve the unresolved-decisions pass and every required design decision. -2. EDIT plan.md with the missing design decisions (interaction state table, empty states, responsive behavior, etc.) and the full required review report. Keep the saved review compact: use the canonical tables and decision IDs. Specify each design requirement once; refer to its section or decision ID from other pass rationales, tasks, and report cells instead of repeating that specification. Give concise score rationales and 10/10 explanations. Retain all required report fields, design decisions, diagrams, ratings, and explanations. +1. Skip the preamble bash block and any AskUserQuestion calls. Read every lazy section the workflow requires: in one response, natively Read plan-design-review/SKILL.md (Read the remainder with offset if the first Read stops early), plan-design-review/sections/review-sections.md (the one lazy section this review requires) and plan.md. No cat, sed, ls, manifest or git exploration is needed. Review all 7 design passes. Rate each scored design dimension 0-10 and explain what would make it a 10; preserve the unresolved-decisions pass and every required design decision. +2. EDIT plan.md with the missing design decisions (interaction state table, empty states, responsive behavior, etc.) and the full required review report. Keep the saved review compact: use the canonical tables and decision IDs. Specify each design requirement once; refer to its section or decision ID from other pass rationales, tasks, and report cells instead of repeating that specification. Give concise score rationales and 10/10 explanations. Retain all required report fields, design decisions, diagrams, ratings, and explanations. Keep everything you add to plan.md, including the report, under 14,000 characters. 3. Persist that complete plan and review with Write before publishing a completed walkthrough or saying a fix is applied. Read plan.md back to verify the saved changes. -4. Then return a brief, concrete summary of the design changes; do not repeat the full review in the response. This changes presentation only: execute every required pass and lazy-section Read. +4. Then return a brief, concrete summary of the design changes in at most ten lines; do not repeat the full review in the response. This changes presentation only: execute every required pass and lazy-section Read. IMPORTANT: Do NOT try to browse any URLs or use a browse binary. This is a plan review, not a live site audit.`, workingDirectory: reviewDir,