From 76f38fcd0d0d0659fe6f34a1717dc8f842f8e471 Mon Sep 17 00:00:00 2001 From: Garry Tan Date: Fri, 28 Aug 2026 02:36:09 +0000 Subject: [PATCH] =?UTF-8?q?test:=20absorb=20the=20ponytail-import=20wave?= =?UTF-8?q?=20into=20the=20guard=20fixtures=20=E2=80=94=20ceilings,=20sche?= =?UTF-8?q?ma=20pin,=20triad=20phrasing?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Skeleton ceilings re-captured for the 17 carved skills the wave deliberately grew (reuse ladder + bounded closer + shortcut trail, net of the gated -236B AUQ cut), each with its measured size in the comment per the carve-guards protocol. eval-store schema pin updated to v2 (harvest gains insertions/deletions/net). The AUQ prose-triad keeps its pinned per-choice phrasing ('explicit on EACH choice') while still deferring the score scale to the canonical Format rule — the shipped cut is strictly closer to the pre-cut text than the render that already passed the NOT-WORSE gate. Autoplan carve anchors follow the Phase 2.5 renumbering. Golden ship baselines re-captured. Co-Authored-By: Claude Fable 5 --- autoplan/SKILL.md | 2 +- canary/SKILL.md | 2 +- codex/SKILL.md | 2 +- context-restore/SKILL.md | 2 +- context-save/SKILL.md | 2 +- cso/SKILL.md | 2 +- design-consultation/SKILL.md | 2 +- design-html/SKILL.md | 2 +- design-review/SKILL.md | 2 +- design-shotgun/SKILL.md | 2 +- devex-review/SKILL.md | 2 +- document-generate/SKILL.md | 2 +- document-release/SKILL.md | 2 +- health/SKILL.md | 2 +- investigate/SKILL.md | 2 +- ios-clean/SKILL.md | 2 +- ios-design-review/SKILL.md | 2 +- ios-fix/SKILL.md | 2 +- ios-qa/SKILL.md | 2 +- ios-sync/SKILL.md | 2 +- land-and-deploy/SKILL.md | 2 +- landing-report/SKILL.md | 2 +- learn/SKILL.md | 2 +- office-hours/SKILL.md | 2 +- pair-agent/SKILL.md | 2 +- plan-ceo-review/SKILL.md | 2 +- plan-design-review/SKILL.md | 2 +- plan-devex-review/SKILL.md | 2 +- plan-eng-review/SKILL.md | 2 +- plan-tune/SKILL.md | 2 +- qa-only/SKILL.md | 2 +- qa/SKILL.md | 2 +- retro/SKILL.md | 2 +- review/SKILL.md | 2 +- .../preamble/generate-ask-user-format.ts | 2 +- setup-deploy/SKILL.md | 2 +- setup-gbrain/SKILL.md | 2 +- ship/SKILL.md | 2 +- skillify/SKILL.md | 2 +- spec/SKILL.md | 2 +- sync-gbrain/SKILL.md | 2 +- test/fixtures/golden/claude-ship-SKILL.md | 2 +- test/fixtures/golden/codex-ship-SKILL.md | 2 +- test/fixtures/golden/factory-ship-SKILL.md | 2 +- test/helpers/carve-guards.ts | 38 +++++++++---------- test/helpers/eval-store.test.ts | 2 +- 46 files changed, 64 insertions(+), 64 deletions(-) diff --git a/autoplan/SKILL.md b/autoplan/SKILL.md index 3dbedcd0f..b6cfbff07 100644 --- a/autoplan/SKILL.md +++ b/autoplan/SKILL.md @@ -98,7 +98,7 @@ Tell three outcomes apart: **Prose fallback — render the decision brief as a markdown message, not a tool call.** Same information as the tool format below, different structure (paragraphs, not ✅/❌ bullets). It MUST surface this triad: 1. **A clear ELI10 of the issue itself** — plain English on what's being decided and why it matters (the question, not per-choice), naming the stakes. Lead with it. -2. **Completeness scores per choice** — per the Completeness rule in the Format section below; never silently drop the score. +2. **Completeness scores per choice** — explicit on EACH choice, per the Completeness rule in the Format section below; never silently drop the score. 3. **The recommendation and why** — the `Recommendation: because ` line plus the `(recommended)` marker on that choice. Layout: a `D` title + a one-line note to reply with a letter (in Conductor this is the normal path; elsewhere it means AskUserQuestion was unavailable or errored); the issue ELI10; the Recommendation line; then ONE paragraph per choice carrying its `(recommended)` marker, its `Completeness: X/10`, and 2-4 sentences of reasoning — never a bare bullet list; a closing `Net:` line. Split chains / 5+ options: one prose block per per-option call, in sequence. Then STOP and wait — the user's typed answer is the decision. In plan mode this satisfies end-of-turn like a tool call. diff --git a/canary/SKILL.md b/canary/SKILL.md index c498fe0d1..980d034c0 100644 --- a/canary/SKILL.md +++ b/canary/SKILL.md @@ -91,7 +91,7 @@ Tell three outcomes apart: **Prose fallback — render the decision brief as a markdown message, not a tool call.** Same information as the tool format below, different structure (paragraphs, not ✅/❌ bullets). It MUST surface this triad: 1. **A clear ELI10 of the issue itself** — plain English on what's being decided and why it matters (the question, not per-choice), naming the stakes. Lead with it. -2. **Completeness scores per choice** — per the Completeness rule in the Format section below; never silently drop the score. +2. **Completeness scores per choice** — explicit on EACH choice, per the Completeness rule in the Format section below; never silently drop the score. 3. **The recommendation and why** — the `Recommendation: because ` line plus the `(recommended)` marker on that choice. Layout: a `D` title + a one-line note to reply with a letter (in Conductor this is the normal path; elsewhere it means AskUserQuestion was unavailable or errored); the issue ELI10; the Recommendation line; then ONE paragraph per choice carrying its `(recommended)` marker, its `Completeness: X/10`, and 2-4 sentences of reasoning — never a bare bullet list; a closing `Net:` line. Split chains / 5+ options: one prose block per per-option call, in sequence. Then STOP and wait — the user's typed answer is the decision. In plan mode this satisfies end-of-turn like a tool call. diff --git a/codex/SKILL.md b/codex/SKILL.md index 521646735..2d95bb175 100644 --- a/codex/SKILL.md +++ b/codex/SKILL.md @@ -94,7 +94,7 @@ Tell three outcomes apart: **Prose fallback — render the decision brief as a markdown message, not a tool call.** Same information as the tool format below, different structure (paragraphs, not ✅/❌ bullets). It MUST surface this triad: 1. **A clear ELI10 of the issue itself** — plain English on what's being decided and why it matters (the question, not per-choice), naming the stakes. Lead with it. -2. **Completeness scores per choice** — per the Completeness rule in the Format section below; never silently drop the score. +2. **Completeness scores per choice** — explicit on EACH choice, per the Completeness rule in the Format section below; never silently drop the score. 3. **The recommendation and why** — the `Recommendation: because ` line plus the `(recommended)` marker on that choice. Layout: a `D` title + a one-line note to reply with a letter (in Conductor this is the normal path; elsewhere it means AskUserQuestion was unavailable or errored); the issue ELI10; the Recommendation line; then ONE paragraph per choice carrying its `(recommended)` marker, its `Completeness: X/10`, and 2-4 sentences of reasoning — never a bare bullet list; a closing `Net:` line. Split chains / 5+ options: one prose block per per-option call, in sequence. Then STOP and wait — the user's typed answer is the decision. In plan mode this satisfies end-of-turn like a tool call. diff --git a/context-restore/SKILL.md b/context-restore/SKILL.md index 0ef4eb43a..cef2d420e 100644 --- a/context-restore/SKILL.md +++ b/context-restore/SKILL.md @@ -95,7 +95,7 @@ Tell three outcomes apart: **Prose fallback — render the decision brief as a markdown message, not a tool call.** Same information as the tool format below, different structure (paragraphs, not ✅/❌ bullets). It MUST surface this triad: 1. **A clear ELI10 of the issue itself** — plain English on what's being decided and why it matters (the question, not per-choice), naming the stakes. Lead with it. -2. **Completeness scores per choice** — per the Completeness rule in the Format section below; never silently drop the score. +2. **Completeness scores per choice** — explicit on EACH choice, per the Completeness rule in the Format section below; never silently drop the score. 3. **The recommendation and why** — the `Recommendation: because ` line plus the `(recommended)` marker on that choice. Layout: a `D` title + a one-line note to reply with a letter (in Conductor this is the normal path; elsewhere it means AskUserQuestion was unavailable or errored); the issue ELI10; the Recommendation line; then ONE paragraph per choice carrying its `(recommended)` marker, its `Completeness: X/10`, and 2-4 sentences of reasoning — never a bare bullet list; a closing `Net:` line. Split chains / 5+ options: one prose block per per-option call, in sequence. Then STOP and wait — the user's typed answer is the decision. In plan mode this satisfies end-of-turn like a tool call. diff --git a/context-save/SKILL.md b/context-save/SKILL.md index 562d9822e..4af2114f0 100644 --- a/context-save/SKILL.md +++ b/context-save/SKILL.md @@ -94,7 +94,7 @@ Tell three outcomes apart: **Prose fallback — render the decision brief as a markdown message, not a tool call.** Same information as the tool format below, different structure (paragraphs, not ✅/❌ bullets). It MUST surface this triad: 1. **A clear ELI10 of the issue itself** — plain English on what's being decided and why it matters (the question, not per-choice), naming the stakes. Lead with it. -2. **Completeness scores per choice** — per the Completeness rule in the Format section below; never silently drop the score. +2. **Completeness scores per choice** — explicit on EACH choice, per the Completeness rule in the Format section below; never silently drop the score. 3. **The recommendation and why** — the `Recommendation: because ` line plus the `(recommended)` marker on that choice. Layout: a `D` title + a one-line note to reply with a letter (in Conductor this is the normal path; elsewhere it means AskUserQuestion was unavailable or errored); the issue ELI10; the Recommendation line; then ONE paragraph per choice carrying its `(recommended)` marker, its `Completeness: X/10`, and 2-4 sentences of reasoning — never a bare bullet list; a closing `Net:` line. Split chains / 5+ options: one prose block per per-option call, in sequence. Then STOP and wait — the user's typed answer is the decision. In plan mode this satisfies end-of-turn like a tool call. diff --git a/cso/SKILL.md b/cso/SKILL.md index 6ff92a920..483f770c6 100644 --- a/cso/SKILL.md +++ b/cso/SKILL.md @@ -97,7 +97,7 @@ Tell three outcomes apart: **Prose fallback — render the decision brief as a markdown message, not a tool call.** Same information as the tool format below, different structure (paragraphs, not ✅/❌ bullets). It MUST surface this triad: 1. **A clear ELI10 of the issue itself** — plain English on what's being decided and why it matters (the question, not per-choice), naming the stakes. Lead with it. -2. **Completeness scores per choice** — per the Completeness rule in the Format section below; never silently drop the score. +2. **Completeness scores per choice** — explicit on EACH choice, per the Completeness rule in the Format section below; never silently drop the score. 3. **The recommendation and why** — the `Recommendation: because ` line plus the `(recommended)` marker on that choice. Layout: a `D` title + a one-line note to reply with a letter (in Conductor this is the normal path; elsewhere it means AskUserQuestion was unavailable or errored); the issue ELI10; the Recommendation line; then ONE paragraph per choice carrying its `(recommended)` marker, its `Completeness: X/10`, and 2-4 sentences of reasoning — never a bare bullet list; a closing `Net:` line. Split chains / 5+ options: one prose block per per-option call, in sequence. Then STOP and wait — the user's typed answer is the decision. In plan mode this satisfies end-of-turn like a tool call. diff --git a/design-consultation/SKILL.md b/design-consultation/SKILL.md index 2db165b10..7a1b10156 100644 --- a/design-consultation/SKILL.md +++ b/design-consultation/SKILL.md @@ -117,7 +117,7 @@ Tell three outcomes apart: **Prose fallback — render the decision brief as a markdown message, not a tool call.** Same information as the tool format below, different structure (paragraphs, not ✅/❌ bullets). It MUST surface this triad: 1. **A clear ELI10 of the issue itself** — plain English on what's being decided and why it matters (the question, not per-choice), naming the stakes. Lead with it. -2. **Completeness scores per choice** — per the Completeness rule in the Format section below; never silently drop the score. +2. **Completeness scores per choice** — explicit on EACH choice, per the Completeness rule in the Format section below; never silently drop the score. 3. **The recommendation and why** — the `Recommendation: because ` line plus the `(recommended)` marker on that choice. Layout: a `D` title + a one-line note to reply with a letter (in Conductor this is the normal path; elsewhere it means AskUserQuestion was unavailable or errored); the issue ELI10; the Recommendation line; then ONE paragraph per choice carrying its `(recommended)` marker, its `Completeness: X/10`, and 2-4 sentences of reasoning — never a bare bullet list; a closing `Net:` line. Split chains / 5+ options: one prose block per per-option call, in sequence. Then STOP and wait — the user's typed answer is the decision. In plan mode this satisfies end-of-turn like a tool call. diff --git a/design-html/SKILL.md b/design-html/SKILL.md index 99acbddb2..0c186af6c 100644 --- a/design-html/SKILL.md +++ b/design-html/SKILL.md @@ -98,7 +98,7 @@ Tell three outcomes apart: **Prose fallback — render the decision brief as a markdown message, not a tool call.** Same information as the tool format below, different structure (paragraphs, not ✅/❌ bullets). It MUST surface this triad: 1. **A clear ELI10 of the issue itself** — plain English on what's being decided and why it matters (the question, not per-choice), naming the stakes. Lead with it. -2. **Completeness scores per choice** — per the Completeness rule in the Format section below; never silently drop the score. +2. **Completeness scores per choice** — explicit on EACH choice, per the Completeness rule in the Format section below; never silently drop the score. 3. **The recommendation and why** — the `Recommendation: because ` line plus the `(recommended)` marker on that choice. Layout: a `D` title + a one-line note to reply with a letter (in Conductor this is the normal path; elsewhere it means AskUserQuestion was unavailable or errored); the issue ELI10; the Recommendation line; then ONE paragraph per choice carrying its `(recommended)` marker, its `Completeness: X/10`, and 2-4 sentences of reasoning — never a bare bullet list; a closing `Net:` line. Split chains / 5+ options: one prose block per per-option call, in sequence. Then STOP and wait — the user's typed answer is the decision. In plan mode this satisfies end-of-turn like a tool call. diff --git a/design-review/SKILL.md b/design-review/SKILL.md index 961bc817c..421ecd2a1 100644 --- a/design-review/SKILL.md +++ b/design-review/SKILL.md @@ -95,7 +95,7 @@ Tell three outcomes apart: **Prose fallback — render the decision brief as a markdown message, not a tool call.** Same information as the tool format below, different structure (paragraphs, not ✅/❌ bullets). It MUST surface this triad: 1. **A clear ELI10 of the issue itself** — plain English on what's being decided and why it matters (the question, not per-choice), naming the stakes. Lead with it. -2. **Completeness scores per choice** — per the Completeness rule in the Format section below; never silently drop the score. +2. **Completeness scores per choice** — explicit on EACH choice, per the Completeness rule in the Format section below; never silently drop the score. 3. **The recommendation and why** — the `Recommendation: because ` line plus the `(recommended)` marker on that choice. Layout: a `D` title + a one-line note to reply with a letter (in Conductor this is the normal path; elsewhere it means AskUserQuestion was unavailable or errored); the issue ELI10; the Recommendation line; then ONE paragraph per choice carrying its `(recommended)` marker, its `Completeness: X/10`, and 2-4 sentences of reasoning — never a bare bullet list; a closing `Net:` line. Split chains / 5+ options: one prose block per per-option call, in sequence. Then STOP and wait — the user's typed answer is the decision. In plan mode this satisfies end-of-turn like a tool call. diff --git a/design-shotgun/SKILL.md b/design-shotgun/SKILL.md index a91601b96..e545d8a04 100644 --- a/design-shotgun/SKILL.md +++ b/design-shotgun/SKILL.md @@ -112,7 +112,7 @@ Tell three outcomes apart: **Prose fallback — render the decision brief as a markdown message, not a tool call.** Same information as the tool format below, different structure (paragraphs, not ✅/❌ bullets). It MUST surface this triad: 1. **A clear ELI10 of the issue itself** — plain English on what's being decided and why it matters (the question, not per-choice), naming the stakes. Lead with it. -2. **Completeness scores per choice** — per the Completeness rule in the Format section below; never silently drop the score. +2. **Completeness scores per choice** — explicit on EACH choice, per the Completeness rule in the Format section below; never silently drop the score. 3. **The recommendation and why** — the `Recommendation: because ` line plus the `(recommended)` marker on that choice. Layout: a `D` title + a one-line note to reply with a letter (in Conductor this is the normal path; elsewhere it means AskUserQuestion was unavailable or errored); the issue ELI10; the Recommendation line; then ONE paragraph per choice carrying its `(recommended)` marker, its `Completeness: X/10`, and 2-4 sentences of reasoning — never a bare bullet list; a closing `Net:` line. Split chains / 5+ options: one prose block per per-option call, in sequence. Then STOP and wait — the user's typed answer is the decision. In plan mode this satisfies end-of-turn like a tool call. diff --git a/devex-review/SKILL.md b/devex-review/SKILL.md index aeffbda15..0b539fe82 100644 --- a/devex-review/SKILL.md +++ b/devex-review/SKILL.md @@ -97,7 +97,7 @@ Tell three outcomes apart: **Prose fallback — render the decision brief as a markdown message, not a tool call.** Same information as the tool format below, different structure (paragraphs, not ✅/❌ bullets). It MUST surface this triad: 1. **A clear ELI10 of the issue itself** — plain English on what's being decided and why it matters (the question, not per-choice), naming the stakes. Lead with it. -2. **Completeness scores per choice** — per the Completeness rule in the Format section below; never silently drop the score. +2. **Completeness scores per choice** — explicit on EACH choice, per the Completeness rule in the Format section below; never silently drop the score. 3. **The recommendation and why** — the `Recommendation: because ` line plus the `(recommended)` marker on that choice. Layout: a `D` title + a one-line note to reply with a letter (in Conductor this is the normal path; elsewhere it means AskUserQuestion was unavailable or errored); the issue ELI10; the Recommendation line; then ONE paragraph per choice carrying its `(recommended)` marker, its `Completeness: X/10`, and 2-4 sentences of reasoning — never a bare bullet list; a closing `Net:` line. Split chains / 5+ options: one prose block per per-option call, in sequence. Then STOP and wait — the user's typed answer is the decision. In plan mode this satisfies end-of-turn like a tool call. diff --git a/document-generate/SKILL.md b/document-generate/SKILL.md index edacc099a..d0c54e8c7 100644 --- a/document-generate/SKILL.md +++ b/document-generate/SKILL.md @@ -97,7 +97,7 @@ Tell three outcomes apart: **Prose fallback — render the decision brief as a markdown message, not a tool call.** Same information as the tool format below, different structure (paragraphs, not ✅/❌ bullets). It MUST surface this triad: 1. **A clear ELI10 of the issue itself** — plain English on what's being decided and why it matters (the question, not per-choice), naming the stakes. Lead with it. -2. **Completeness scores per choice** — per the Completeness rule in the Format section below; never silently drop the score. +2. **Completeness scores per choice** — explicit on EACH choice, per the Completeness rule in the Format section below; never silently drop the score. 3. **The recommendation and why** — the `Recommendation: because ` line plus the `(recommended)` marker on that choice. Layout: a `D` title + a one-line note to reply with a letter (in Conductor this is the normal path; elsewhere it means AskUserQuestion was unavailable or errored); the issue ELI10; the Recommendation line; then ONE paragraph per choice carrying its `(recommended)` marker, its `Completeness: X/10`, and 2-4 sentences of reasoning — never a bare bullet list; a closing `Net:` line. Split chains / 5+ options: one prose block per per-option call, in sequence. Then STOP and wait — the user's typed answer is the decision. In plan mode this satisfies end-of-turn like a tool call. diff --git a/document-release/SKILL.md b/document-release/SKILL.md index 46cf8e16c..7936ade24 100644 --- a/document-release/SKILL.md +++ b/document-release/SKILL.md @@ -95,7 +95,7 @@ Tell three outcomes apart: **Prose fallback — render the decision brief as a markdown message, not a tool call.** Same information as the tool format below, different structure (paragraphs, not ✅/❌ bullets). It MUST surface this triad: 1. **A clear ELI10 of the issue itself** — plain English on what's being decided and why it matters (the question, not per-choice), naming the stakes. Lead with it. -2. **Completeness scores per choice** — per the Completeness rule in the Format section below; never silently drop the score. +2. **Completeness scores per choice** — explicit on EACH choice, per the Completeness rule in the Format section below; never silently drop the score. 3. **The recommendation and why** — the `Recommendation: because ` line plus the `(recommended)` marker on that choice. Layout: a `D` title + a one-line note to reply with a letter (in Conductor this is the normal path; elsewhere it means AskUserQuestion was unavailable or errored); the issue ELI10; the Recommendation line; then ONE paragraph per choice carrying its `(recommended)` marker, its `Completeness: X/10`, and 2-4 sentences of reasoning — never a bare bullet list; a closing `Net:` line. Split chains / 5+ options: one prose block per per-option call, in sequence. Then STOP and wait — the user's typed answer is the decision. In plan mode this satisfies end-of-turn like a tool call. diff --git a/health/SKILL.md b/health/SKILL.md index 84b66beac..3027c67e3 100644 --- a/health/SKILL.md +++ b/health/SKILL.md @@ -93,7 +93,7 @@ Tell three outcomes apart: **Prose fallback — render the decision brief as a markdown message, not a tool call.** Same information as the tool format below, different structure (paragraphs, not ✅/❌ bullets). It MUST surface this triad: 1. **A clear ELI10 of the issue itself** — plain English on what's being decided and why it matters (the question, not per-choice), naming the stakes. Lead with it. -2. **Completeness scores per choice** — per the Completeness rule in the Format section below; never silently drop the score. +2. **Completeness scores per choice** — explicit on EACH choice, per the Completeness rule in the Format section below; never silently drop the score. 3. **The recommendation and why** — the `Recommendation: because ` line plus the `(recommended)` marker on that choice. Layout: a `D` title + a one-line note to reply with a letter (in Conductor this is the normal path; elsewhere it means AskUserQuestion was unavailable or errored); the issue ELI10; the Recommendation line; then ONE paragraph per choice carrying its `(recommended)` marker, its `Completeness: X/10`, and 2-4 sentences of reasoning — never a bare bullet list; a closing `Net:` line. Split chains / 5+ options: one prose block per per-option call, in sequence. Then STOP and wait — the user's typed answer is the decision. In plan mode this satisfies end-of-turn like a tool call. diff --git a/investigate/SKILL.md b/investigate/SKILL.md index 0f847c14a..65d4a283f 100644 --- a/investigate/SKILL.md +++ b/investigate/SKILL.md @@ -132,7 +132,7 @@ Tell three outcomes apart: **Prose fallback — render the decision brief as a markdown message, not a tool call.** Same information as the tool format below, different structure (paragraphs, not ✅/❌ bullets). It MUST surface this triad: 1. **A clear ELI10 of the issue itself** — plain English on what's being decided and why it matters (the question, not per-choice), naming the stakes. Lead with it. -2. **Completeness scores per choice** — per the Completeness rule in the Format section below; never silently drop the score. +2. **Completeness scores per choice** — explicit on EACH choice, per the Completeness rule in the Format section below; never silently drop the score. 3. **The recommendation and why** — the `Recommendation: because ` line plus the `(recommended)` marker on that choice. Layout: a `D` title + a one-line note to reply with a letter (in Conductor this is the normal path; elsewhere it means AskUserQuestion was unavailable or errored); the issue ELI10; the Recommendation line; then ONE paragraph per choice carrying its `(recommended)` marker, its `Completeness: X/10`, and 2-4 sentences of reasoning — never a bare bullet list; a closing `Net:` line. Split chains / 5+ options: one prose block per per-option call, in sequence. Then STOP and wait — the user's typed answer is the decision. In plan mode this satisfies end-of-turn like a tool call. diff --git a/ios-clean/SKILL.md b/ios-clean/SKILL.md index 4decf0668..fe3df2977 100644 --- a/ios-clean/SKILL.md +++ b/ios-clean/SKILL.md @@ -95,7 +95,7 @@ Tell three outcomes apart: **Prose fallback — render the decision brief as a markdown message, not a tool call.** Same information as the tool format below, different structure (paragraphs, not ✅/❌ bullets). It MUST surface this triad: 1. **A clear ELI10 of the issue itself** — plain English on what's being decided and why it matters (the question, not per-choice), naming the stakes. Lead with it. -2. **Completeness scores per choice** — per the Completeness rule in the Format section below; never silently drop the score. +2. **Completeness scores per choice** — explicit on EACH choice, per the Completeness rule in the Format section below; never silently drop the score. 3. **The recommendation and why** — the `Recommendation: because ` line plus the `(recommended)` marker on that choice. Layout: a `D` title + a one-line note to reply with a letter (in Conductor this is the normal path; elsewhere it means AskUserQuestion was unavailable or errored); the issue ELI10; the Recommendation line; then ONE paragraph per choice carrying its `(recommended)` marker, its `Completeness: X/10`, and 2-4 sentences of reasoning — never a bare bullet list; a closing `Net:` line. Split chains / 5+ options: one prose block per per-option call, in sequence. Then STOP and wait — the user's typed answer is the decision. In plan mode this satisfies end-of-turn like a tool call. diff --git a/ios-design-review/SKILL.md b/ios-design-review/SKILL.md index 17e53472b..93c73fa35 100644 --- a/ios-design-review/SKILL.md +++ b/ios-design-review/SKILL.md @@ -97,7 +97,7 @@ Tell three outcomes apart: **Prose fallback — render the decision brief as a markdown message, not a tool call.** Same information as the tool format below, different structure (paragraphs, not ✅/❌ bullets). It MUST surface this triad: 1. **A clear ELI10 of the issue itself** — plain English on what's being decided and why it matters (the question, not per-choice), naming the stakes. Lead with it. -2. **Completeness scores per choice** — per the Completeness rule in the Format section below; never silently drop the score. +2. **Completeness scores per choice** — explicit on EACH choice, per the Completeness rule in the Format section below; never silently drop the score. 3. **The recommendation and why** — the `Recommendation: because ` line plus the `(recommended)` marker on that choice. Layout: a `D` title + a one-line note to reply with a letter (in Conductor this is the normal path; elsewhere it means AskUserQuestion was unavailable or errored); the issue ELI10; the Recommendation line; then ONE paragraph per choice carrying its `(recommended)` marker, its `Completeness: X/10`, and 2-4 sentences of reasoning — never a bare bullet list; a closing `Net:` line. Split chains / 5+ options: one prose block per per-option call, in sequence. Then STOP and wait — the user's typed answer is the decision. In plan mode this satisfies end-of-turn like a tool call. diff --git a/ios-fix/SKILL.md b/ios-fix/SKILL.md index 60d2e9516..c2d78c683 100644 --- a/ios-fix/SKILL.md +++ b/ios-fix/SKILL.md @@ -98,7 +98,7 @@ Tell three outcomes apart: **Prose fallback — render the decision brief as a markdown message, not a tool call.** Same information as the tool format below, different structure (paragraphs, not ✅/❌ bullets). It MUST surface this triad: 1. **A clear ELI10 of the issue itself** — plain English on what's being decided and why it matters (the question, not per-choice), naming the stakes. Lead with it. -2. **Completeness scores per choice** — per the Completeness rule in the Format section below; never silently drop the score. +2. **Completeness scores per choice** — explicit on EACH choice, per the Completeness rule in the Format section below; never silently drop the score. 3. **The recommendation and why** — the `Recommendation: because ` line plus the `(recommended)` marker on that choice. Layout: a `D` title + a one-line note to reply with a letter (in Conductor this is the normal path; elsewhere it means AskUserQuestion was unavailable or errored); the issue ELI10; the Recommendation line; then ONE paragraph per choice carrying its `(recommended)` marker, its `Completeness: X/10`, and 2-4 sentences of reasoning — never a bare bullet list; a closing `Net:` line. Split chains / 5+ options: one prose block per per-option call, in sequence. Then STOP and wait — the user's typed answer is the decision. In plan mode this satisfies end-of-turn like a tool call. diff --git a/ios-qa/SKILL.md b/ios-qa/SKILL.md index c57ae82fe..380dac958 100644 --- a/ios-qa/SKILL.md +++ b/ios-qa/SKILL.md @@ -101,7 +101,7 @@ Tell three outcomes apart: **Prose fallback — render the decision brief as a markdown message, not a tool call.** Same information as the tool format below, different structure (paragraphs, not ✅/❌ bullets). It MUST surface this triad: 1. **A clear ELI10 of the issue itself** — plain English on what's being decided and why it matters (the question, not per-choice), naming the stakes. Lead with it. -2. **Completeness scores per choice** — per the Completeness rule in the Format section below; never silently drop the score. +2. **Completeness scores per choice** — explicit on EACH choice, per the Completeness rule in the Format section below; never silently drop the score. 3. **The recommendation and why** — the `Recommendation: because ` line plus the `(recommended)` marker on that choice. Layout: a `D` title + a one-line note to reply with a letter (in Conductor this is the normal path; elsewhere it means AskUserQuestion was unavailable or errored); the issue ELI10; the Recommendation line; then ONE paragraph per choice carrying its `(recommended)` marker, its `Completeness: X/10`, and 2-4 sentences of reasoning — never a bare bullet list; a closing `Net:` line. Split chains / 5+ options: one prose block per per-option call, in sequence. Then STOP and wait — the user's typed answer is the decision. In plan mode this satisfies end-of-turn like a tool call. diff --git a/ios-sync/SKILL.md b/ios-sync/SKILL.md index 0f0dbfa55..0bc5798e6 100644 --- a/ios-sync/SKILL.md +++ b/ios-sync/SKILL.md @@ -95,7 +95,7 @@ Tell three outcomes apart: **Prose fallback — render the decision brief as a markdown message, not a tool call.** Same information as the tool format below, different structure (paragraphs, not ✅/❌ bullets). It MUST surface this triad: 1. **A clear ELI10 of the issue itself** — plain English on what's being decided and why it matters (the question, not per-choice), naming the stakes. Lead with it. -2. **Completeness scores per choice** — per the Completeness rule in the Format section below; never silently drop the score. +2. **Completeness scores per choice** — explicit on EACH choice, per the Completeness rule in the Format section below; never silently drop the score. 3. **The recommendation and why** — the `Recommendation: because ` line plus the `(recommended)` marker on that choice. Layout: a `D` title + a one-line note to reply with a letter (in Conductor this is the normal path; elsewhere it means AskUserQuestion was unavailable or errored); the issue ELI10; the Recommendation line; then ONE paragraph per choice carrying its `(recommended)` marker, its `Completeness: X/10`, and 2-4 sentences of reasoning — never a bare bullet list; a closing `Net:` line. Split chains / 5+ options: one prose block per per-option call, in sequence. Then STOP and wait — the user's typed answer is the decision. In plan mode this satisfies end-of-turn like a tool call. diff --git a/land-and-deploy/SKILL.md b/land-and-deploy/SKILL.md index da7b67ce0..0708d2884 100644 --- a/land-and-deploy/SKILL.md +++ b/land-and-deploy/SKILL.md @@ -90,7 +90,7 @@ Tell three outcomes apart: **Prose fallback — render the decision brief as a markdown message, not a tool call.** Same information as the tool format below, different structure (paragraphs, not ✅/❌ bullets). It MUST surface this triad: 1. **A clear ELI10 of the issue itself** — plain English on what's being decided and why it matters (the question, not per-choice), naming the stakes. Lead with it. -2. **Completeness scores per choice** — per the Completeness rule in the Format section below; never silently drop the score. +2. **Completeness scores per choice** — explicit on EACH choice, per the Completeness rule in the Format section below; never silently drop the score. 3. **The recommendation and why** — the `Recommendation: because ` line plus the `(recommended)` marker on that choice. Layout: a `D` title + a one-line note to reply with a letter (in Conductor this is the normal path; elsewhere it means AskUserQuestion was unavailable or errored); the issue ELI10; the Recommendation line; then ONE paragraph per choice carrying its `(recommended)` marker, its `Completeness: X/10`, and 2-4 sentences of reasoning — never a bare bullet list; a closing `Net:` line. Split chains / 5+ options: one prose block per per-option call, in sequence. Then STOP and wait — the user's typed answer is the decision. In plan mode this satisfies end-of-turn like a tool call. diff --git a/landing-report/SKILL.md b/landing-report/SKILL.md index 2a56c2dc0..b432b568b 100644 --- a/landing-report/SKILL.md +++ b/landing-report/SKILL.md @@ -92,7 +92,7 @@ Tell three outcomes apart: **Prose fallback — render the decision brief as a markdown message, not a tool call.** Same information as the tool format below, different structure (paragraphs, not ✅/❌ bullets). It MUST surface this triad: 1. **A clear ELI10 of the issue itself** — plain English on what's being decided and why it matters (the question, not per-choice), naming the stakes. Lead with it. -2. **Completeness scores per choice** — per the Completeness rule in the Format section below; never silently drop the score. +2. **Completeness scores per choice** — explicit on EACH choice, per the Completeness rule in the Format section below; never silently drop the score. 3. **The recommendation and why** — the `Recommendation: because ` line plus the `(recommended)` marker on that choice. Layout: a `D` title + a one-line note to reply with a letter (in Conductor this is the normal path; elsewhere it means AskUserQuestion was unavailable or errored); the issue ELI10; the Recommendation line; then ONE paragraph per choice carrying its `(recommended)` marker, its `Completeness: X/10`, and 2-4 sentences of reasoning — never a bare bullet list; a closing `Net:` line. Split chains / 5+ options: one prose block per per-option call, in sequence. Then STOP and wait — the user's typed answer is the decision. In plan mode this satisfies end-of-turn like a tool call. diff --git a/learn/SKILL.md b/learn/SKILL.md index a368a8a1a..66d2fbc6d 100644 --- a/learn/SKILL.md +++ b/learn/SKILL.md @@ -93,7 +93,7 @@ Tell three outcomes apart: **Prose fallback — render the decision brief as a markdown message, not a tool call.** Same information as the tool format below, different structure (paragraphs, not ✅/❌ bullets). It MUST surface this triad: 1. **A clear ELI10 of the issue itself** — plain English on what's being decided and why it matters (the question, not per-choice), naming the stakes. Lead with it. -2. **Completeness scores per choice** — per the Completeness rule in the Format section below; never silently drop the score. +2. **Completeness scores per choice** — explicit on EACH choice, per the Completeness rule in the Format section below; never silently drop the score. 3. **The recommendation and why** — the `Recommendation: because ` line plus the `(recommended)` marker on that choice. Layout: a `D` title + a one-line note to reply with a letter (in Conductor this is the normal path; elsewhere it means AskUserQuestion was unavailable or errored); the issue ELI10; the Recommendation line; then ONE paragraph per choice carrying its `(recommended)` marker, its `Completeness: X/10`, and 2-4 sentences of reasoning — never a bare bullet list; a closing `Net:` line. Split chains / 5+ options: one prose block per per-option call, in sequence. Then STOP and wait — the user's typed answer is the decision. In plan mode this satisfies end-of-turn like a tool call. diff --git a/office-hours/SKILL.md b/office-hours/SKILL.md index 26e604bb3..a68be8103 100644 --- a/office-hours/SKILL.md +++ b/office-hours/SKILL.md @@ -128,7 +128,7 @@ Tell three outcomes apart: **Prose fallback — render the decision brief as a markdown message, not a tool call.** Same information as the tool format below, different structure (paragraphs, not ✅/❌ bullets). It MUST surface this triad: 1. **A clear ELI10 of the issue itself** — plain English on what's being decided and why it matters (the question, not per-choice), naming the stakes. Lead with it. -2. **Completeness scores per choice** — per the Completeness rule in the Format section below; never silently drop the score. +2. **Completeness scores per choice** — explicit on EACH choice, per the Completeness rule in the Format section below; never silently drop the score. 3. **The recommendation and why** — the `Recommendation: because ` line plus the `(recommended)` marker on that choice. Layout: a `D` title + a one-line note to reply with a letter (in Conductor this is the normal path; elsewhere it means AskUserQuestion was unavailable or errored); the issue ELI10; the Recommendation line; then ONE paragraph per choice carrying its `(recommended)` marker, its `Completeness: X/10`, and 2-4 sentences of reasoning — never a bare bullet list; a closing `Net:` line. Split chains / 5+ options: one prose block per per-option call, in sequence. Then STOP and wait — the user's typed answer is the decision. In plan mode this satisfies end-of-turn like a tool call. diff --git a/pair-agent/SKILL.md b/pair-agent/SKILL.md index fd19c82a0..0e911557a 100644 --- a/pair-agent/SKILL.md +++ b/pair-agent/SKILL.md @@ -94,7 +94,7 @@ Tell three outcomes apart: **Prose fallback — render the decision brief as a markdown message, not a tool call.** Same information as the tool format below, different structure (paragraphs, not ✅/❌ bullets). It MUST surface this triad: 1. **A clear ELI10 of the issue itself** — plain English on what's being decided and why it matters (the question, not per-choice), naming the stakes. Lead with it. -2. **Completeness scores per choice** — per the Completeness rule in the Format section below; never silently drop the score. +2. **Completeness scores per choice** — explicit on EACH choice, per the Completeness rule in the Format section below; never silently drop the score. 3. **The recommendation and why** — the `Recommendation: because ` line plus the `(recommended)` marker on that choice. Layout: a `D` title + a one-line note to reply with a letter (in Conductor this is the normal path; elsewhere it means AskUserQuestion was unavailable or errored); the issue ELI10; the Recommendation line; then ONE paragraph per choice carrying its `(recommended)` marker, its `Completeness: X/10`, and 2-4 sentences of reasoning — never a bare bullet list; a closing `Net:` line. Split chains / 5+ options: one prose block per per-option call, in sequence. Then STOP and wait — the user's typed answer is the decision. In plan mode this satisfies end-of-turn like a tool call. diff --git a/plan-ceo-review/SKILL.md b/plan-ceo-review/SKILL.md index 8c3a492dd..5df11757a 100644 --- a/plan-ceo-review/SKILL.md +++ b/plan-ceo-review/SKILL.md @@ -120,7 +120,7 @@ Tell three outcomes apart: **Prose fallback — render the decision brief as a markdown message, not a tool call.** Same information as the tool format below, different structure (paragraphs, not ✅/❌ bullets). It MUST surface this triad: 1. **A clear ELI10 of the issue itself** — plain English on what's being decided and why it matters (the question, not per-choice), naming the stakes. Lead with it. -2. **Completeness scores per choice** — per the Completeness rule in the Format section below; never silently drop the score. +2. **Completeness scores per choice** — explicit on EACH choice, per the Completeness rule in the Format section below; never silently drop the score. 3. **The recommendation and why** — the `Recommendation: because ` line plus the `(recommended)` marker on that choice. Layout: a `D` title + a one-line note to reply with a letter (in Conductor this is the normal path; elsewhere it means AskUserQuestion was unavailable or errored); the issue ELI10; the Recommendation line; then ONE paragraph per choice carrying its `(recommended)` marker, its `Completeness: X/10`, and 2-4 sentences of reasoning — never a bare bullet list; a closing `Net:` line. Split chains / 5+ options: one prose block per per-option call, in sequence. Then STOP and wait — the user's typed answer is the decision. In plan mode this satisfies end-of-turn like a tool call. diff --git a/plan-design-review/SKILL.md b/plan-design-review/SKILL.md index c69defb39..b2173ac0e 100644 --- a/plan-design-review/SKILL.md +++ b/plan-design-review/SKILL.md @@ -93,7 +93,7 @@ Tell three outcomes apart: **Prose fallback — render the decision brief as a markdown message, not a tool call.** Same information as the tool format below, different structure (paragraphs, not ✅/❌ bullets). It MUST surface this triad: 1. **A clear ELI10 of the issue itself** — plain English on what's being decided and why it matters (the question, not per-choice), naming the stakes. Lead with it. -2. **Completeness scores per choice** — per the Completeness rule in the Format section below; never silently drop the score. +2. **Completeness scores per choice** — explicit on EACH choice, per the Completeness rule in the Format section below; never silently drop the score. 3. **The recommendation and why** — the `Recommendation: because ` line plus the `(recommended)` marker on that choice. Layout: a `D` title + a one-line note to reply with a letter (in Conductor this is the normal path; elsewhere it means AskUserQuestion was unavailable or errored); the issue ELI10; the Recommendation line; then ONE paragraph per choice carrying its `(recommended)` marker, its `Completeness: X/10`, and 2-4 sentences of reasoning — never a bare bullet list; a closing `Net:` line. Split chains / 5+ options: one prose block per per-option call, in sequence. Then STOP and wait — the user's typed answer is the decision. In plan mode this satisfies end-of-turn like a tool call. diff --git a/plan-devex-review/SKILL.md b/plan-devex-review/SKILL.md index ef5282cd8..022855e7e 100644 --- a/plan-devex-review/SKILL.md +++ b/plan-devex-review/SKILL.md @@ -98,7 +98,7 @@ Tell three outcomes apart: **Prose fallback — render the decision brief as a markdown message, not a tool call.** Same information as the tool format below, different structure (paragraphs, not ✅/❌ bullets). It MUST surface this triad: 1. **A clear ELI10 of the issue itself** — plain English on what's being decided and why it matters (the question, not per-choice), naming the stakes. Lead with it. -2. **Completeness scores per choice** — per the Completeness rule in the Format section below; never silently drop the score. +2. **Completeness scores per choice** — explicit on EACH choice, per the Completeness rule in the Format section below; never silently drop the score. 3. **The recommendation and why** — the `Recommendation: because ` line plus the `(recommended)` marker on that choice. Layout: a `D` title + a one-line note to reply with a letter (in Conductor this is the normal path; elsewhere it means AskUserQuestion was unavailable or errored); the issue ELI10; the Recommendation line; then ONE paragraph per choice carrying its `(recommended)` marker, its `Completeness: X/10`, and 2-4 sentences of reasoning — never a bare bullet list; a closing `Net:` line. Split chains / 5+ options: one prose block per per-option call, in sequence. Then STOP and wait — the user's typed answer is the decision. In plan mode this satisfies end-of-turn like a tool call. diff --git a/plan-eng-review/SKILL.md b/plan-eng-review/SKILL.md index c95709881..38aefc1f3 100644 --- a/plan-eng-review/SKILL.md +++ b/plan-eng-review/SKILL.md @@ -96,7 +96,7 @@ Tell three outcomes apart: **Prose fallback — render the decision brief as a markdown message, not a tool call.** Same information as the tool format below, different structure (paragraphs, not ✅/❌ bullets). It MUST surface this triad: 1. **A clear ELI10 of the issue itself** — plain English on what's being decided and why it matters (the question, not per-choice), naming the stakes. Lead with it. -2. **Completeness scores per choice** — per the Completeness rule in the Format section below; never silently drop the score. +2. **Completeness scores per choice** — explicit on EACH choice, per the Completeness rule in the Format section below; never silently drop the score. 3. **The recommendation and why** — the `Recommendation: because ` line plus the `(recommended)` marker on that choice. Layout: a `D` title + a one-line note to reply with a letter (in Conductor this is the normal path; elsewhere it means AskUserQuestion was unavailable or errored); the issue ELI10; the Recommendation line; then ONE paragraph per choice carrying its `(recommended)` marker, its `Completeness: X/10`, and 2-4 sentences of reasoning — never a bare bullet list; a closing `Net:` line. Split chains / 5+ options: one prose block per per-option call, in sequence. Then STOP and wait — the user's typed answer is the decision. In plan mode this satisfies end-of-turn like a tool call. diff --git a/plan-tune/SKILL.md b/plan-tune/SKILL.md index 116e1e105..825601e11 100644 --- a/plan-tune/SKILL.md +++ b/plan-tune/SKILL.md @@ -103,7 +103,7 @@ Tell three outcomes apart: **Prose fallback — render the decision brief as a markdown message, not a tool call.** Same information as the tool format below, different structure (paragraphs, not ✅/❌ bullets). It MUST surface this triad: 1. **A clear ELI10 of the issue itself** — plain English on what's being decided and why it matters (the question, not per-choice), naming the stakes. Lead with it. -2. **Completeness scores per choice** — per the Completeness rule in the Format section below; never silently drop the score. +2. **Completeness scores per choice** — explicit on EACH choice, per the Completeness rule in the Format section below; never silently drop the score. 3. **The recommendation and why** — the `Recommendation: because ` line plus the `(recommended)` marker on that choice. Layout: a `D` title + a one-line note to reply with a letter (in Conductor this is the normal path; elsewhere it means AskUserQuestion was unavailable or errored); the issue ELI10; the Recommendation line; then ONE paragraph per choice carrying its `(recommended)` marker, its `Completeness: X/10`, and 2-4 sentences of reasoning — never a bare bullet list; a closing `Net:` line. Split chains / 5+ options: one prose block per per-option call, in sequence. Then STOP and wait — the user's typed answer is the decision. In plan mode this satisfies end-of-turn like a tool call. diff --git a/qa-only/SKILL.md b/qa-only/SKILL.md index 61fb87f9c..0896f1534 100644 --- a/qa-only/SKILL.md +++ b/qa-only/SKILL.md @@ -93,7 +93,7 @@ Tell three outcomes apart: **Prose fallback — render the decision brief as a markdown message, not a tool call.** Same information as the tool format below, different structure (paragraphs, not ✅/❌ bullets). It MUST surface this triad: 1. **A clear ELI10 of the issue itself** — plain English on what's being decided and why it matters (the question, not per-choice), naming the stakes. Lead with it. -2. **Completeness scores per choice** — per the Completeness rule in the Format section below; never silently drop the score. +2. **Completeness scores per choice** — explicit on EACH choice, per the Completeness rule in the Format section below; never silently drop the score. 3. **The recommendation and why** — the `Recommendation: because ` line plus the `(recommended)` marker on that choice. Layout: a `D` title + a one-line note to reply with a letter (in Conductor this is the normal path; elsewhere it means AskUserQuestion was unavailable or errored); the issue ELI10; the Recommendation line; then ONE paragraph per choice carrying its `(recommended)` marker, its `Completeness: X/10`, and 2-4 sentences of reasoning — never a bare bullet list; a closing `Net:` line. Split chains / 5+ options: one prose block per per-option call, in sequence. Then STOP and wait — the user's typed answer is the decision. In plan mode this satisfies end-of-turn like a tool call. diff --git a/qa/SKILL.md b/qa/SKILL.md index b77890d1c..6b7bbfd02 100644 --- a/qa/SKILL.md +++ b/qa/SKILL.md @@ -99,7 +99,7 @@ Tell three outcomes apart: **Prose fallback — render the decision brief as a markdown message, not a tool call.** Same information as the tool format below, different structure (paragraphs, not ✅/❌ bullets). It MUST surface this triad: 1. **A clear ELI10 of the issue itself** — plain English on what's being decided and why it matters (the question, not per-choice), naming the stakes. Lead with it. -2. **Completeness scores per choice** — per the Completeness rule in the Format section below; never silently drop the score. +2. **Completeness scores per choice** — explicit on EACH choice, per the Completeness rule in the Format section below; never silently drop the score. 3. **The recommendation and why** — the `Recommendation: because ` line plus the `(recommended)` marker on that choice. Layout: a `D` title + a one-line note to reply with a letter (in Conductor this is the normal path; elsewhere it means AskUserQuestion was unavailable or errored); the issue ELI10; the Recommendation line; then ONE paragraph per choice carrying its `(recommended)` marker, its `Completeness: X/10`, and 2-4 sentences of reasoning — never a bare bullet list; a closing `Net:` line. Split chains / 5+ options: one prose block per per-option call, in sequence. Then STOP and wait — the user's typed answer is the decision. In plan mode this satisfies end-of-turn like a tool call. diff --git a/retro/SKILL.md b/retro/SKILL.md index d0f37eb87..ce1cfaa8a 100644 --- a/retro/SKILL.md +++ b/retro/SKILL.md @@ -113,7 +113,7 @@ Tell three outcomes apart: **Prose fallback — render the decision brief as a markdown message, not a tool call.** Same information as the tool format below, different structure (paragraphs, not ✅/❌ bullets). It MUST surface this triad: 1. **A clear ELI10 of the issue itself** — plain English on what's being decided and why it matters (the question, not per-choice), naming the stakes. Lead with it. -2. **Completeness scores per choice** — per the Completeness rule in the Format section below; never silently drop the score. +2. **Completeness scores per choice** — explicit on EACH choice, per the Completeness rule in the Format section below; never silently drop the score. 3. **The recommendation and why** — the `Recommendation: because ` line plus the `(recommended)` marker on that choice. Layout: a `D` title + a one-line note to reply with a letter (in Conductor this is the normal path; elsewhere it means AskUserQuestion was unavailable or errored); the issue ELI10; the Recommendation line; then ONE paragraph per choice carrying its `(recommended)` marker, its `Completeness: X/10`, and 2-4 sentences of reasoning — never a bare bullet list; a closing `Net:` line. Split chains / 5+ options: one prose block per per-option call, in sequence. Then STOP and wait — the user's typed answer is the decision. In plan mode this satisfies end-of-turn like a tool call. diff --git a/review/SKILL.md b/review/SKILL.md index b596b7ae1..8ca078428 100644 --- a/review/SKILL.md +++ b/review/SKILL.md @@ -95,7 +95,7 @@ Tell three outcomes apart: **Prose fallback — render the decision brief as a markdown message, not a tool call.** Same information as the tool format below, different structure (paragraphs, not ✅/❌ bullets). It MUST surface this triad: 1. **A clear ELI10 of the issue itself** — plain English on what's being decided and why it matters (the question, not per-choice), naming the stakes. Lead with it. -2. **Completeness scores per choice** — per the Completeness rule in the Format section below; never silently drop the score. +2. **Completeness scores per choice** — explicit on EACH choice, per the Completeness rule in the Format section below; never silently drop the score. 3. **The recommendation and why** — the `Recommendation: because ` line plus the `(recommended)` marker on that choice. Layout: a `D` title + a one-line note to reply with a letter (in Conductor this is the normal path; elsewhere it means AskUserQuestion was unavailable or errored); the issue ELI10; the Recommendation line; then ONE paragraph per choice carrying its `(recommended)` marker, its `Completeness: X/10`, and 2-4 sentences of reasoning — never a bare bullet list; a closing `Net:` line. Split chains / 5+ options: one prose block per per-option call, in sequence. Then STOP and wait — the user's typed answer is the decision. In plan mode this satisfies end-of-turn like a tool call. diff --git a/scripts/resolvers/preamble/generate-ask-user-format.ts b/scripts/resolvers/preamble/generate-ask-user-format.ts index 8b54849ad..d24df7d9a 100644 --- a/scripts/resolvers/preamble/generate-ask-user-format.ts +++ b/scripts/resolvers/preamble/generate-ask-user-format.ts @@ -26,7 +26,7 @@ Tell three outcomes apart: **Prose fallback — render the decision brief as a markdown message, not a tool call.** Same information as the tool format below, different structure (paragraphs, not ✅/❌ bullets). It MUST surface this triad: 1. **A clear ELI10 of the issue itself** — plain English on what's being decided and why it matters (the question, not per-choice), naming the stakes. Lead with it. -2. **Completeness scores per choice** — per the Completeness rule in the Format section below; never silently drop the score. +2. **Completeness scores per choice** — explicit on EACH choice, per the Completeness rule in the Format section below; never silently drop the score. 3. **The recommendation and why** — the \`Recommendation: because \` line plus the \`(recommended)\` marker on that choice. Layout: a \`D\` title + a one-line note to reply with a letter (in Conductor this is the normal path; elsewhere it means AskUserQuestion was unavailable or errored); the issue ELI10; the Recommendation line; then ONE paragraph per choice carrying its \`(recommended)\` marker, its \`Completeness: X/10\`, and 2-4 sentences of reasoning — never a bare bullet list; a closing \`Net:\` line. Split chains / 5+ options: one prose block per per-option call, in sequence. Then STOP and wait — the user's typed answer is the decision. In plan mode this satisfies end-of-turn like a tool call. diff --git a/setup-deploy/SKILL.md b/setup-deploy/SKILL.md index 9cf1c228f..8d3dd061b 100644 --- a/setup-deploy/SKILL.md +++ b/setup-deploy/SKILL.md @@ -94,7 +94,7 @@ Tell three outcomes apart: **Prose fallback — render the decision brief as a markdown message, not a tool call.** Same information as the tool format below, different structure (paragraphs, not ✅/❌ bullets). It MUST surface this triad: 1. **A clear ELI10 of the issue itself** — plain English on what's being decided and why it matters (the question, not per-choice), naming the stakes. Lead with it. -2. **Completeness scores per choice** — per the Completeness rule in the Format section below; never silently drop the score. +2. **Completeness scores per choice** — explicit on EACH choice, per the Completeness rule in the Format section below; never silently drop the score. 3. **The recommendation and why** — the `Recommendation: because ` line plus the `(recommended)` marker on that choice. Layout: a `D` title + a one-line note to reply with a letter (in Conductor this is the normal path; elsewhere it means AskUserQuestion was unavailable or errored); the issue ELI10; the Recommendation line; then ONE paragraph per choice carrying its `(recommended)` marker, its `Completeness: X/10`, and 2-4 sentences of reasoning — never a bare bullet list; a closing `Net:` line. Split chains / 5+ options: one prose block per per-option call, in sequence. Then STOP and wait — the user's typed answer is the decision. In plan mode this satisfies end-of-turn like a tool call. diff --git a/setup-gbrain/SKILL.md b/setup-gbrain/SKILL.md index 9399021d7..f56bc8d4a 100644 --- a/setup-gbrain/SKILL.md +++ b/setup-gbrain/SKILL.md @@ -93,7 +93,7 @@ Tell three outcomes apart: **Prose fallback — render the decision brief as a markdown message, not a tool call.** Same information as the tool format below, different structure (paragraphs, not ✅/❌ bullets). It MUST surface this triad: 1. **A clear ELI10 of the issue itself** — plain English on what's being decided and why it matters (the question, not per-choice), naming the stakes. Lead with it. -2. **Completeness scores per choice** — per the Completeness rule in the Format section below; never silently drop the score. +2. **Completeness scores per choice** — explicit on EACH choice, per the Completeness rule in the Format section below; never silently drop the score. 3. **The recommendation and why** — the `Recommendation: because ` line plus the `(recommended)` marker on that choice. Layout: a `D` title + a one-line note to reply with a letter (in Conductor this is the normal path; elsewhere it means AskUserQuestion was unavailable or errored); the issue ELI10; the Recommendation line; then ONE paragraph per choice carrying its `(recommended)` marker, its `Completeness: X/10`, and 2-4 sentences of reasoning — never a bare bullet list; a closing `Net:` line. Split chains / 5+ options: one prose block per per-option call, in sequence. Then STOP and wait — the user's typed answer is the decision. In plan mode this satisfies end-of-turn like a tool call. diff --git a/ship/SKILL.md b/ship/SKILL.md index 1ad2f3ad6..c132054f0 100644 --- a/ship/SKILL.md +++ b/ship/SKILL.md @@ -95,7 +95,7 @@ Tell three outcomes apart: **Prose fallback — render the decision brief as a markdown message, not a tool call.** Same information as the tool format below, different structure (paragraphs, not ✅/❌ bullets). It MUST surface this triad: 1. **A clear ELI10 of the issue itself** — plain English on what's being decided and why it matters (the question, not per-choice), naming the stakes. Lead with it. -2. **Completeness scores per choice** — per the Completeness rule in the Format section below; never silently drop the score. +2. **Completeness scores per choice** — explicit on EACH choice, per the Completeness rule in the Format section below; never silently drop the score. 3. **The recommendation and why** — the `Recommendation: because ` line plus the `(recommended)` marker on that choice. Layout: a `D` title + a one-line note to reply with a letter (in Conductor this is the normal path; elsewhere it means AskUserQuestion was unavailable or errored); the issue ELI10; the Recommendation line; then ONE paragraph per choice carrying its `(recommended)` marker, its `Completeness: X/10`, and 2-4 sentences of reasoning — never a bare bullet list; a closing `Net:` line. Split chains / 5+ options: one prose block per per-option call, in sequence. Then STOP and wait — the user's typed answer is the decision. In plan mode this satisfies end-of-turn like a tool call. diff --git a/skillify/SKILL.md b/skillify/SKILL.md index a996e3c03..0dd9330f0 100644 --- a/skillify/SKILL.md +++ b/skillify/SKILL.md @@ -92,7 +92,7 @@ Tell three outcomes apart: **Prose fallback — render the decision brief as a markdown message, not a tool call.** Same information as the tool format below, different structure (paragraphs, not ✅/❌ bullets). It MUST surface this triad: 1. **A clear ELI10 of the issue itself** — plain English on what's being decided and why it matters (the question, not per-choice), naming the stakes. Lead with it. -2. **Completeness scores per choice** — per the Completeness rule in the Format section below; never silently drop the score. +2. **Completeness scores per choice** — explicit on EACH choice, per the Completeness rule in the Format section below; never silently drop the score. 3. **The recommendation and why** — the `Recommendation: because ` line plus the `(recommended)` marker on that choice. Layout: a `D` title + a one-line note to reply with a letter (in Conductor this is the normal path; elsewhere it means AskUserQuestion was unavailable or errored); the issue ELI10; the Recommendation line; then ONE paragraph per choice carrying its `(recommended)` marker, its `Completeness: X/10`, and 2-4 sentences of reasoning — never a bare bullet list; a closing `Net:` line. Split chains / 5+ options: one prose block per per-option call, in sequence. Then STOP and wait — the user's typed answer is the decision. In plan mode this satisfies end-of-turn like a tool call. diff --git a/spec/SKILL.md b/spec/SKILL.md index 1b4291aa4..7707bfb33 100644 --- a/spec/SKILL.md +++ b/spec/SKILL.md @@ -93,7 +93,7 @@ Tell three outcomes apart: **Prose fallback — render the decision brief as a markdown message, not a tool call.** Same information as the tool format below, different structure (paragraphs, not ✅/❌ bullets). It MUST surface this triad: 1. **A clear ELI10 of the issue itself** — plain English on what's being decided and why it matters (the question, not per-choice), naming the stakes. Lead with it. -2. **Completeness scores per choice** — per the Completeness rule in the Format section below; never silently drop the score. +2. **Completeness scores per choice** — explicit on EACH choice, per the Completeness rule in the Format section below; never silently drop the score. 3. **The recommendation and why** — the `Recommendation: because ` line plus the `(recommended)` marker on that choice. Layout: a `D` title + a one-line note to reply with a letter (in Conductor this is the normal path; elsewhere it means AskUserQuestion was unavailable or errored); the issue ELI10; the Recommendation line; then ONE paragraph per choice carrying its `(recommended)` marker, its `Completeness: X/10`, and 2-4 sentences of reasoning — never a bare bullet list; a closing `Net:` line. Split chains / 5+ options: one prose block per per-option call, in sequence. Then STOP and wait — the user's typed answer is the decision. In plan mode this satisfies end-of-turn like a tool call. diff --git a/sync-gbrain/SKILL.md b/sync-gbrain/SKILL.md index 3ef975b0c..38a7fb1d2 100644 --- a/sync-gbrain/SKILL.md +++ b/sync-gbrain/SKILL.md @@ -94,7 +94,7 @@ Tell three outcomes apart: **Prose fallback — render the decision brief as a markdown message, not a tool call.** Same information as the tool format below, different structure (paragraphs, not ✅/❌ bullets). It MUST surface this triad: 1. **A clear ELI10 of the issue itself** — plain English on what's being decided and why it matters (the question, not per-choice), naming the stakes. Lead with it. -2. **Completeness scores per choice** — per the Completeness rule in the Format section below; never silently drop the score. +2. **Completeness scores per choice** — explicit on EACH choice, per the Completeness rule in the Format section below; never silently drop the score. 3. **The recommendation and why** — the `Recommendation: because ` line plus the `(recommended)` marker on that choice. Layout: a `D` title + a one-line note to reply with a letter (in Conductor this is the normal path; elsewhere it means AskUserQuestion was unavailable or errored); the issue ELI10; the Recommendation line; then ONE paragraph per choice carrying its `(recommended)` marker, its `Completeness: X/10`, and 2-4 sentences of reasoning — never a bare bullet list; a closing `Net:` line. Split chains / 5+ options: one prose block per per-option call, in sequence. Then STOP and wait — the user's typed answer is the decision. In plan mode this satisfies end-of-turn like a tool call. diff --git a/test/fixtures/golden/claude-ship-SKILL.md b/test/fixtures/golden/claude-ship-SKILL.md index 1ad2f3ad6..c132054f0 100644 --- a/test/fixtures/golden/claude-ship-SKILL.md +++ b/test/fixtures/golden/claude-ship-SKILL.md @@ -95,7 +95,7 @@ Tell three outcomes apart: **Prose fallback — render the decision brief as a markdown message, not a tool call.** Same information as the tool format below, different structure (paragraphs, not ✅/❌ bullets). It MUST surface this triad: 1. **A clear ELI10 of the issue itself** — plain English on what's being decided and why it matters (the question, not per-choice), naming the stakes. Lead with it. -2. **Completeness scores per choice** — per the Completeness rule in the Format section below; never silently drop the score. +2. **Completeness scores per choice** — explicit on EACH choice, per the Completeness rule in the Format section below; never silently drop the score. 3. **The recommendation and why** — the `Recommendation: because ` line plus the `(recommended)` marker on that choice. Layout: a `D` title + a one-line note to reply with a letter (in Conductor this is the normal path; elsewhere it means AskUserQuestion was unavailable or errored); the issue ELI10; the Recommendation line; then ONE paragraph per choice carrying its `(recommended)` marker, its `Completeness: X/10`, and 2-4 sentences of reasoning — never a bare bullet list; a closing `Net:` line. Split chains / 5+ options: one prose block per per-option call, in sequence. Then STOP and wait — the user's typed answer is the decision. In plan mode this satisfies end-of-turn like a tool call. diff --git a/test/fixtures/golden/codex-ship-SKILL.md b/test/fixtures/golden/codex-ship-SKILL.md index 84a3dab9e..8abe197cc 100644 --- a/test/fixtures/golden/codex-ship-SKILL.md +++ b/test/fixtures/golden/codex-ship-SKILL.md @@ -81,7 +81,7 @@ Tell three outcomes apart: **Prose fallback — render the decision brief as a markdown message, not a tool call.** Same information as the tool format below, different structure (paragraphs, not ✅/❌ bullets). It MUST surface this triad: 1. **A clear ELI10 of the issue itself** — plain English on what's being decided and why it matters (the question, not per-choice), naming the stakes. Lead with it. -2. **Completeness scores per choice** — per the Completeness rule in the Format section below; never silently drop the score. +2. **Completeness scores per choice** — explicit on EACH choice, per the Completeness rule in the Format section below; never silently drop the score. 3. **The recommendation and why** — the `Recommendation: because ` line plus the `(recommended)` marker on that choice. Layout: a `D` title + a one-line note to reply with a letter (in Conductor this is the normal path; elsewhere it means AskUserQuestion was unavailable or errored); the issue ELI10; the Recommendation line; then ONE paragraph per choice carrying its `(recommended)` marker, its `Completeness: X/10`, and 2-4 sentences of reasoning — never a bare bullet list; a closing `Net:` line. Split chains / 5+ options: one prose block per per-option call, in sequence. Then STOP and wait — the user's typed answer is the decision. In plan mode this satisfies end-of-turn like a tool call. diff --git a/test/fixtures/golden/factory-ship-SKILL.md b/test/fixtures/golden/factory-ship-SKILL.md index efe7c5b48..f0f6f6476 100644 --- a/test/fixtures/golden/factory-ship-SKILL.md +++ b/test/fixtures/golden/factory-ship-SKILL.md @@ -83,7 +83,7 @@ Tell three outcomes apart: **Prose fallback — render the decision brief as a markdown message, not a tool call.** Same information as the tool format below, different structure (paragraphs, not ✅/❌ bullets). It MUST surface this triad: 1. **A clear ELI10 of the issue itself** — plain English on what's being decided and why it matters (the question, not per-choice), naming the stakes. Lead with it. -2. **Completeness scores per choice** — per the Completeness rule in the Format section below; never silently drop the score. +2. **Completeness scores per choice** — explicit on EACH choice, per the Completeness rule in the Format section below; never silently drop the score. 3. **The recommendation and why** — the `Recommendation: because ` line plus the `(recommended)` marker on that choice. Layout: a `D` title + a one-line note to reply with a letter (in Conductor this is the normal path; elsewhere it means AskUserQuestion was unavailable or errored); the issue ELI10; the Recommendation line; then ONE paragraph per choice carrying its `(recommended)` marker, its `Completeness: X/10`, and 2-4 sentences of reasoning — never a bare bullet list; a closing `Net:` line. Split chains / 5+ options: one prose block per per-option call, in sequence. Then STOP and wait — the user's typed answer is the decision. In plan mode this satisfies end-of-turn like a tool call. diff --git a/test/helpers/carve-guards.ts b/test/helpers/carve-guards.ts index e27998ad6..f4641f95c 100644 --- a/test/helpers/carve-guards.ts +++ b/test/helpers/carve-guards.ts @@ -150,7 +150,7 @@ export const CARVE_GUARDS: Record = { }, behavioral: 'external', externalTest: 'test/skill-e2e-ship-section-loading.test.ts', - maxSkeletonBytes: 71_300, // token-reduction Phases 1-2 + #2700 document-release anchors; measured 70,568 post-merge regen + maxSkeletonBytes: 73_090, // ponytail-import wave: reuse ladder + bounded closer + shortcut trail (AUQ repetition cut netted -236B, gated); measured 72_482 minUnionBytes: 181_000, // token-reduction Phases 1-2 (v1.69.x branch); measured union 201,464 mustContain: ['VERSION', 'CHANGELOG', 'review', 'merge', 'PR'], // v1.58.5.0: pre-push-guard install (#2077) stacks on the shared first-run-guidance preamble. @@ -181,7 +181,7 @@ export const CARVE_GUARDS: Record = { // v1.65 merge: provisional larger-of-both-waves budget; re-measured below. // Fork port wave 2 (#703): the repo-doc-preference block in the design // check grew every plan-review skeleton ~0.7KB. Measured values noted. - maxSkeletonBytes: 73_980, // token-reduction Phases 1-2 (v1.69.x branch): preamble bash -> bin/gstack-skill-start, onboarding -> gated emission; measured 73,381 + maxSkeletonBytes: 74_830, // ponytail-import wave: reuse ladder + bounded closer + shortcut trail (AUQ repetition cut netted -236B, gated); measured 74_221 minUnionBytes: 123_600, // token-reduction Phases 1-2 (v1.69.x branch): preamble bash -> bin/gstack-skill-start, onboarding -> gated emission; measured union 137,346 mustContain: ['SCOPE EXPANSION', 'SELECTIVE EXPANSION', 'HOLD SCOPE', 'SCOPE REDUCTION'], // Default-on Codex outside-voice (codexPreflight block + CODEX_MODE branch @@ -207,7 +207,7 @@ export const CARVE_GUARDS: Record = { // check grew every plan-review skeleton ~0.7KB. Measured values noted. // #2499 project-scope MCP jq in the brain-sync block grew every tier-2+ // skeleton ~1.5KB (entry resolution emitted once per SKILL.md). - maxSkeletonBytes: 51_860, // token-reduction Phases 1-2 (v1.69.x branch): preamble bash -> bin/gstack-skill-start, onboarding -> gated emission; measured 51,264 + maxSkeletonBytes: 52_710, // ponytail-import wave: reuse ladder + bounded closer + shortcut trail (AUQ repetition cut netted -236B, gated); measured 52_104 minUnionBytes: 99_800, // token-reduction Phases 1-2 (v1.69.x branch); measured union 110,910 mustContain: ['Architecture', 'Code Quality', 'Test', 'Performance'], // Cross-cutting preamble growth (v1.57.2.0 AUQ-failure prose fallback + the @@ -240,7 +240,7 @@ export const CARVE_GUARDS: Record = { // tier-2+ skeleton (measured 89,184). Main's v1.64.0.0 adds ~340 B more // (telemetry --error-message/--failed-step preamble prose, PR #769). // Budget covers the sum of both waves. - maxSkeletonBytes: 71_840, // token-reduction Phases 1-2 (v1.69.x branch): preamble bash -> bin/gstack-skill-start, onboarding -> gated emission; measured 71,242 + maxSkeletonBytes: 72_690, // ponytail-import wave: reuse ladder + bounded closer + shortcut trail (AUQ repetition cut netted -236B, gated); measured 72_082 minUnionBytes: 99_200, // token-reduction Phases 1-2 (v1.69.x branch); measured union 110,293 mustContain: ['design', 'visual'], maxSizeRatio: 1.12, // D1 1.104 + main's ~0.008 @@ -264,7 +264,7 @@ export const CARVE_GUARDS: Record = { // check grew every plan-review skeleton ~0.7KB. Measured values noted. // #2499 project-scope MCP jq in the brain-sync block grew every tier-2+ // skeleton ~1.5KB (entry resolution emitted once per SKILL.md). - maxSkeletonBytes: 63_580, // token-reduction Phases 1-2 (v1.69.x branch): preamble bash -> bin/gstack-skill-start, onboarding -> gated emission; measured 62,977 + maxSkeletonBytes: 64_420, // ponytail-import wave: reuse ladder + bounded closer + shortcut trail (AUQ repetition cut netted -236B, gated); measured 63_817 minUnionBytes: 99_700, // token-reduction Phases 1-2 (v1.69.x branch); measured union 110,833 mustContain: ['developer experience', 'Getting Started'], // Default-on Codex outside-voice (codexPreflight block + CODEX_MODE branch @@ -295,7 +295,7 @@ export const CARVE_GUARDS: Record = { // the #538 opt-out + D1 evidence directive — ratio 1.104 measured. // #2499 project-scope MCP jq in the brain-sync block grew every tier-2+ // skeleton ~1.5KB (entry resolution emitted once per SKILL.md). - maxSkeletonBytes: 68_200, // token-reduction Phase 4 wave 4 (v1.69.x branch): 2A/2B carved out; measured 66,852 + maxSkeletonBytes: 69_500, // ponytail-import wave: reuse ladder + bounded closer + shortcut trail (AUQ repetition cut netted -236B, gated); measured 68_895 minUnionBytes: 115_800, // Phase 4 wave 4; measured union 118,175 mustContain: ['design doc', 'problem statement'], maxSizeRatio: 1.12, @@ -347,7 +347,7 @@ export const CARVE_GUARDS: Record = { // v1.65 merge: provisional larger-of-both-waves budget; re-measured below. // v1.64.1.0: shared-preamble prose from the two parallel v1.64 waves lands // the skeleton at 69,022 B; +~1 KB headroom. - maxSkeletonBytes: 51_500, // token-reduction Phases 1-2 (v1.69.x branch): preamble bash -> bin/gstack-skill-start, onboarding -> gated emission; measured 50,899 + maxSkeletonBytes: 52_340, // ponytail-import wave: reuse ladder + bounded closer + shortcut trail (AUQ repetition cut netted -236B, gated); measured 51_739 minUnionBytes: 65_000, // token-reduction Phases 1-2 (v1.69.x branch): preamble bash -> bin/gstack-skill-start, onboarding -> gated emission; measured union 72,252 mustContain: ['Typography', 'Color', 'Aesthetic Direction'], // Cross-cutting preamble growth (v1.57.2.0 AUQ-failure prose fallback ~2KB + @@ -424,7 +424,7 @@ export const CARVE_GUARDS: Record = { gateAfterStop: undefined, // operational multi-STOP skill, like ship }, behavioral: 'plan', - maxSkeletonBytes: 55_600, // Phase 4 wave 1; measured 55,010 + maxSkeletonBytes: 57_660, // ponytail-import wave: reuse ladder + bounded closer + shortcut trail (AUQ repetition cut netted -236B, gated); measured 57_053 minUnionBytes: 89_000, // Phase 4 wave 1; measured union 93,357 mustContain: ['confidence', 'P1', 'P2', 'Review Army', 'adversarial'], }, @@ -451,7 +451,7 @@ export const CARVE_GUARDS: Record = { gateAfterStop: 'EXIT PLAN MODE GATE', }, behavioral: 'prompt', - maxSkeletonBytes: 55_760, // Phase 4 wave 1; measured 55,155 + maxSkeletonBytes: 57_800, // ponytail-import wave: reuse ladder + bounded closer + shortcut trail (AUQ repetition cut netted -236B, gated); measured 57_198 minUnionBytes: 83_400, // Phase 4 wave 1; measured union 84,304 mustContain: ['GATE: PASS', 'CROSS-MODEL ANALYSIS', 'codex exec resume', 'sandbox_mode="read-only"', 'mktemp'], maxSizeRatio: 1.06, // measured 1.040 vs the v1.64.1.0 parity baseline @@ -477,7 +477,7 @@ export const CARVE_GUARDS: Record = { gateAfterStop: undefined, // operational skill }, behavioral: 'prompt', - maxSkeletonBytes: 57_500, // Phase 4 wave 1; estimated ~56.2KB rendered — re-measured at regen + maxSkeletonBytes: 58_370, // ponytail-import wave: reuse ladder + bounded closer + shortcut trail (AUQ repetition cut netted -236B, gated); measured 57_761 minUnionBytes: 91_000, // Phase 4 wave 1; estimated union ~94.9KB mustContain: ['readiness', 'merge', 'canary', 'revert', 'staging'], }, @@ -487,7 +487,7 @@ export const CARVE_GUARDS: Record = { expectedSections: ['ceo-phase.md', 'design-phase.md', 'eng-phase.md', 'dx-phase.md', 'tasks-aggregator.md'], requiredReads: ['ceo-phase.md', 'eng-phase.md', 'tasks-aggregator.md'], scenario: - 'Run the /autoplan pipeline against the plan in PLAN.md. Codex and subagent tools are unavailable — note both voices unavailable (single-reviewer mode) and keep going. The plan has no UI scope and no developer-facing scope, so Phase 2 and Phase 3.5 are skipped (do not read their sections). Execute Phase 1 (CEO) and Phase 3 (Eng) at full depth, run the Phase 4 aggregator step, and produce the Final Approval Gate summary as the report.', + 'Run the /autoplan pipeline against the plan in PLAN.md. Codex and subagent tools are unavailable — note both voices unavailable (single-reviewer mode) and keep going. The plan has no UI scope and no developer-facing scope, so Phase 2 and Phase 2.5 are skipped (do not read their sections). Execute Phase 1 (CEO) and Phase 3 (Eng) at full depth, run the Phase 4 aggregator step, and produce the Final Approval Gate summary as the report.', staticInvariants: { mustStayInSkeleton: [ '## The 6 Decision Principles', @@ -497,7 +497,7 @@ export const CARVE_GUARDS: Record = { '## Phase 0.5: Codex auth + version preflight', '## Pre-Gate Verification', '## Phase 2: Design Review (conditional — skip if no UI scope)', - '## Phase 3.5: DX Review (conditional — skip if no developer-facing scope)', + '## Phase 2.5: DX Review (conditional — skip if no developer-facing scope)', '- Scope gate (the plan under review is already the target)', ], mustPrecedeStop: ['## The 6 Decision Principles', '## Sequential Execution — MANDATORY', '## Decision Classification'], @@ -512,7 +512,7 @@ export const CARVE_GUARDS: Record = { }, behavioral: 'external', externalTest: 'test/skill-e2e-autoplan-chain.test.ts', // phase-complete markers live ONLY in sections — its assertions ARE section-read proof - maxSkeletonBytes: 59_300, // Phase 4 wave 2; measured 58,696 + maxSkeletonBytes: 62_610, // ponytail-import wave: reuse ladder + bounded closer + shortcut trail (AUQ repetition cut netted -236B, gated); measured 62_006 minUnionBytes: 85_000, // measured union 86,926 mustContain: ['6 Decision Principles', 'TASTE DECISION', 'USER CHALLENGE', 'consensus', 'Restore Point'], }, @@ -541,7 +541,7 @@ export const CARVE_GUARDS: Record = { gateAfterStop: undefined, }, behavioral: 'prompt', - maxSkeletonBytes: 51_200, // Phase 4 wave 2: Phases 4.5-5 carved at the post-confirmation boundary; measured 50,681 + maxSkeletonBytes: 53_330, // ponytail-import wave: reuse ladder + bounded closer + shortcut trail (AUQ repetition cut netted -236B, gated); measured 52_724 minUnionBytes: 64_500, // measured union 67,430 mustContain: ['HARD GATE', 'dedupe', 'quality gate', 'acceptance criteria', 'archive'], }, @@ -570,7 +570,7 @@ export const CARVE_GUARDS: Record = { gateAfterStop: undefined, }, behavioral: 'prompt', - maxSkeletonBytes: 57_600, // Phase 4 wave 2; measured 56,954 + maxSkeletonBytes: 58_950, // ponytail-import wave: reuse ladder + bounded closer + shortcut trail (AUQ repetition cut netted -236B, gated); measured 58_344 minUnionBytes: 78_300, // measured union 79,139 mustContain: ['PGLite', 'Supabase', 'claude mcp add', 'read_secret_to_env', 'pooler'], maxSizeRatio: 1.07, // measured 1.051 vs the branch monolith: index + stubs + 4 STOP pointers @@ -605,7 +605,7 @@ export const CARVE_GUARDS: Record = { gateAfterStop: undefined, }, behavioral: 'prompt', - maxSkeletonBytes: 48_750, // Phase 4 wave 3; measured 48,151 + maxSkeletonBytes: 50_800, // ponytail-import wave: reuse ladder + bounded closer + shortcut trail (AUQ repetition cut netted -236B, gated); measured 50_194 minUnionBytes: 69_500, // measured union 70,385 mustContain: ['bug', 'browse', 'fix', 'Health Score Rubric', 'regression'], }, @@ -642,7 +642,7 @@ export const CARVE_GUARDS: Record = { gateAfterStop: undefined, }, behavioral: 'prompt', - maxSkeletonBytes: 69_500, // Phase 4 wave 3; measured 68,483 (script absorption -5.4KB) + maxSkeletonBytes: 71_620, // ponytail-import wave: reuse ladder + bounded closer + shortcut trail (AUQ repetition cut netted -236B, gated); measured 71_020 (retro also gained the Step 11.5 shortcut-debt harvest) minUnionBytes: 66_000, // measured union 73,496 mustContain: ['retrospective', '45-minute gap', 'Ship of the week', 'Praise'], }, @@ -674,7 +674,7 @@ export const CARVE_GUARDS: Record = { gateAfterStop: undefined, // operational skill, no plan-mode gate }, behavioral: 'prompt', - maxSkeletonBytes: 49_900, // Phase 4 wave 4; measured 48,886 + maxSkeletonBytes: 51_140, // ponytail-import wave: reuse ladder + bounded closer + shortcut trail (AUQ repetition cut netted -236B, gated); measured 50_536 minUnionBytes: 57_500, // Phase 4 wave 4; measured union 58,682 mustContain: ["Don't make me think", "Users scan, they don't read", 'The Goodwill Reservoir', 'PRETEXT API CHEATSHEET', 'Pattern 3: Text around obstacles'], }, @@ -701,7 +701,7 @@ export const CARVE_GUARDS: Record = { gateAfterStop: undefined, }, behavioral: 'prompt', - maxSkeletonBytes: 50_600, // Phase 4 wave 4; measured 49,578 + maxSkeletonBytes: 51_850, // ponytail-import wave: reuse ladder + bounded closer + shortcut trail (AUQ repetition cut netted -236B, gated); measured 51_248 minUnionBytes: 53_200, // Phase 4 wave 4; measured union 54,290 mustContain: ["Don't make me think", "Users scan, they don't read", 'trunk test', '44px minimum'], }, diff --git a/test/helpers/eval-store.test.ts b/test/helpers/eval-store.test.ts index 540143aad..c0769225b 100644 --- a/test/helpers/eval-store.test.ts +++ b/test/helpers/eval-store.test.ts @@ -105,7 +105,7 @@ describe('EvalCollector', () => { const filepath = await collector.finalize(); const data: EvalResult = JSON.parse(fs.readFileSync(filepath, 'utf-8')); - expect(data.schema_version).toBe(1); + expect(data.schema_version).toBe(2); expect(data.tier).toBe('e2e'); expect(data.total_tests).toBe(2); expect(data.passed).toBe(1);