diff --git a/evals/parity/transcripts/policy-units.json b/evals/parity/transcripts/policy-units.json index eec2438fe..6afea0e7c 100644 --- a/evals/parity/transcripts/policy-units.json +++ b/evals/parity/transcripts/policy-units.json @@ -43,7 +43,7 @@ "prompt_sha256": "a0fcff9cad68f9305da861fa5a58990e1ef51d9d84298fe2df7ba8a2f56b1dba", "semantic_attempt_sha256": "7724c3d954f421753c597364c383bd76bc01ba9803f99fb14488287bb925aeb6" }, - "policy_sha256": "951fd945e7ecb438bcedadccd563df430652d35f4cff1d513dfdcf16ee77d02d", + "policy_sha256": "c7df25a16073d69da15243f258a2ef72a69c03ee3114eac2c19cf1c7e126532b", "policy_present": true, "prompt_is_not_authority_input": true, "verdict": "PASS" @@ -88,7 +88,7 @@ "prompt_sha256": "04d82a21358f188cb823dcceeecaebea0b4ed8b9c1f8d00220ef3b499c48b1c8", "semantic_attempt_sha256": "9639406995955284517acb30f56b37dedc42a6c08142163dfa8fe0295105cbbf" }, - "policy_sha256": "e5fc3019062512de84ade87c8c45b34c8f46b95a65ef84a49fd25bbdee63e41e", + "policy_sha256": "8dd78f6f8e01eb9cf1fe0f3f43d7fcd072a4f3f2f0dc14d9f1b5d3ff1250b972", "policy_present": true, "prompt_is_not_authority_input": true, "verdict": "PASS" @@ -138,7 +138,7 @@ "prompt_sha256": "8ad639f708a2ccd5c7c438b1dedf6afacdfa4eac3883a7a046ea23f29314e1c4", "semantic_attempt_sha256": "4c450039e4c622fbeaa34b7f7dfc810d3b7369aaf7e5f4d240cd6cc18a31331a" }, - "policy_sha256": "78df79ecb49744ff13f044b8f5a486705ef43172c4dc19d7dbc616d84f4d578c", + "policy_sha256": "7c7ccb7ff64ea6c73b668848ae5a663606798abc345d7e5c6bb0880732bb29c3", "policy_present": true, "prompt_is_not_authority_input": true, "verdict": "PASS" @@ -188,7 +188,7 @@ "prompt_sha256": "d2aa9aee06a116bf04cb62772e8abdf8dbd606811b6d38565389c88e86c79ede", "semantic_attempt_sha256": "64885a66bdc6688b880cbb9a59babf314834c068eb040657501bf287bfcb0d2b" }, - "policy_sha256": "951fd945e7ecb438bcedadccd563df430652d35f4cff1d513dfdcf16ee77d02d", + "policy_sha256": "c7df25a16073d69da15243f258a2ef72a69c03ee3114eac2c19cf1c7e126532b", "policy_present": true, "prompt_is_not_authority_input": true, "verdict": "PASS" @@ -232,7 +232,7 @@ "prompt_sha256": "377e3a2dfc7b04272f857d27e77c9aa3c9f36c3feb63c03e5aee3ecd3dea8e2a", "semantic_attempt_sha256": "a49b1528091886749278dc22e7da4556c090800989b2bca9c0e1c09dd880fed7" }, - "policy_sha256": "7a8cb7346851f9d948139dbf33cb8c6dd5eb8b2cc7de94e30a53824a94f027a6", + "policy_sha256": "7969ac784f398b09cec2ad44f623de95f1e326721043272b12aaec1c622e840c", "policy_present": true, "prompt_is_not_authority_input": true, "verdict": "PASS" @@ -279,7 +279,7 @@ "prompt_sha256": "a01bb977de84e53d8ce3dfa427bcc93d73c6e449cd4e9c1ff63b436fd41fb0d1", "semantic_attempt_sha256": "ce5c64c62c3ff9b60949f34717dffbc908f62ce30f8a4a48ec182eba4363a206" }, - "policy_sha256": "a65efa55d992f63d0bdb356e1e8c4900b966364af1e22f0b6412e20e7ef30a58", + "policy_sha256": "4134088a9b22fd7d3d2d6492c812ecbc55650df25ab1101189dd24721fe1763d", "policy_present": true, "prompt_is_not_authority_input": true, "verdict": "PASS" @@ -330,7 +330,7 @@ "prompt_sha256": "95e97e26268ad7e509527e0f54c943ed6c4105d919f250648a2ad59ce77cbdb9", "semantic_attempt_sha256": "dacd11a78e32aeb6c0065496bedd4a9770f7bbac1036e003d3768fac41eb84f0" }, - "policy_sha256": "951fd945e7ecb438bcedadccd563df430652d35f4cff1d513dfdcf16ee77d02d", + "policy_sha256": "c7df25a16073d69da15243f258a2ef72a69c03ee3114eac2c19cf1c7e126532b", "policy_present": true, "prompt_is_not_authority_input": true, "verdict": "PASS" @@ -381,7 +381,7 @@ "prompt_sha256": "bdd7e8adfa7c15cf8531f84c3adaaacc725075f1a78f7487225300750547b82d", "semantic_attempt_sha256": "32c11b63412c5d873ff29dcc87c45bef1b50daa218a5c6c1075864aa24d7c3b6" }, - "policy_sha256": "951fd945e7ecb438bcedadccd563df430652d35f4cff1d513dfdcf16ee77d02d", + "policy_sha256": "c7df25a16073d69da15243f258a2ef72a69c03ee3114eac2c19cf1c7e126532b", "policy_present": true, "prompt_is_not_authority_input": true, "verdict": "PASS" @@ -426,7 +426,7 @@ "prompt_sha256": "51b7d53bc342e8632e34ae31a167cbde568ab5ee2a6dd2ccc980583a30604164", "semantic_attempt_sha256": "06f77b27c9bd360562655b23ca00a0dd021bdb14ebaaf3d836845e4e0cb43d68" }, - "policy_sha256": "65f28f454b67d4904c1857a2a6cfcff387c0aa6e4ae58aec490bc95289dce59f", + "policy_sha256": "43d5cbbf88ed4f9afa3e7af214b4aa8c3a764a7006d76a864344df05d1f76392", "policy_present": true, "prompt_is_not_authority_input": true, "verdict": "PASS" diff --git a/scripts/gstack2/execution-profiles.ts b/scripts/gstack2/execution-profiles.ts new file mode 100644 index 000000000..b6cff9d69 --- /dev/null +++ b/scripts/gstack2/execution-profiles.ts @@ -0,0 +1,77 @@ +import type { ExecutionProfile } from './types'; + +export interface ExecutionProfileContract { + profile: ExecutionProfile; + inferWhen: string; + mandatoryModules: string; + legalSkips: string; + artifacts: string; + allowedClaims: string; +} + +export const EXECUTION_PROFILE_CONTRACTS: Record = { + readiness: { + profile: 'readiness', + inferWhen: 'A narrow, reversible, pre-deployment or operational readiness decision needs bounded evidence; the requested claim is readiness, not completeness.', + mandatoryModules: 'Every selected specialist module remains mandatory. Run its entry gates and the smallest source-authorized evidence path that can answer the readiness question.', + legalSkips: 'Only source-declared conditional work whose condition is demonstrably false, or evidence unavailable after a named attempt. Never skip a STOP gate, approval boundary, reproduction/root-cause gate, or required physical-device/production evidence.', + artifacts: 'A readiness record naming the exact scope, probes run, evidence and freshness, failures, skipped work with reasons, and the next standard/deep step.', + allowedClaims: 'Only ready/not-ready for the named bounded decision. Must say “Readiness profile — not a complete review.” Never claim comprehensive, fully verified, production-safe, or no issues found outside the inspected scope.', + }, + standard: { + profile: 'standard', + inferWhen: 'Normal feature or change work has bounded scope and risk and needs the selected specialist’s complete default workflow.', + mandatoryModules: 'Every selected specialist module and all of its mandatory phases, gates, artifacts, and exit checks.', + legalSkips: 'Only smart skips explicitly authorized by the specialist and supported by inspected evidence; list each skipped primary module and each skipped conditional phase.', + artifacts: 'Every artifact required by the selected specialist, plus evidence provenance, unresolved decisions, and a skip ledger.', + allowedClaims: 'Complete only for the named specialist scope and evidence layer. Broader product, security, production, or device claims require those modules and evidence.', + }, + deep: { + profile: 'deep', + inferWhen: 'Risk, ambiguity, blast radius, cross-system effects, irreversible mutation, security/reliability needs, or production deployment demands stronger evidence.', + mandatoryModules: 'Every selected specialist module, all mandatory phases and outside-voice/cross-consumer modules selected by the dispatcher, with unchanged STOP and approval gates.', + legalSkips: 'Only specialist-authorized smart skips proven irrelevant. Missing, stale, malformed, or contradictory evidence is a gap or blocker, never a successful skip.', + artifacts: 'All specialist artifacts plus changed-input/unchanged-consumer trace, negative and failure-path evidence, provenance/freshness, rollback or reversibility evidence where applicable, and unresolved-risk ledger.', + allowedClaims: 'Complete only across the explicitly listed modules and evidence layers. “Confirmed” requires independent supporting evidence; production/device/security claims require matching production/device/security evidence.', + }, +}; + +/** Infer from structured operating conditions, never prompt text. */ +export function inferExecutionProfile( + signals: Record, + specialistDefault: ExecutionProfile, +): ExecutionProfile { + const highRisk = signals.risk === 'high' + || signals.blast_radius === 'broad' + || signals.irreversible === true + || signals.audit_focus === 'security' + || signals.audit_focus === 'deep' + || signals.evidence_need === 'independent' + || signals.failure_impact === 'critical' + || signals.mutation_scope === 'consequential' + || signals.external_mutation_authorized === true + || signals.deployment_state === 'production' + || signals.release_stage === 'approved-pr' + || signals.release_stage === 'landed'; + if (highRisk) return 'deep'; + + const boundedReadiness = signals.evidence_need === 'readiness' + && signals.scope === 'narrow' + && signals.irreversible !== true + && signals.mutation_scope !== 'consequential' + && signals.external_mutation_authorized !== true + && signals.deployment_state !== 'production' + && signals.release_stage !== 'approved-pr' + && signals.release_stage !== 'landed'; + if (boundedReadiness) return 'readiness'; + + return specialistDefault; +} + +export function renderExecutionProfiles(): string { + const rows = (['readiness', 'standard', 'deep'] as const).map((name) => { + const contract = EXECUTION_PROFILE_CONTRACTS[name]; + return `## ${name === 'readiness' ? 'Smoke/readiness' : name[0].toUpperCase() + name.slice(1)}\n\n- Infer when: ${contract.inferWhen}\n- Mandatory modules: ${contract.mandatoryModules}\n- Legal skips: ${contract.legalSkips}\n- Artifacts: ${contract.artifacts}\n- Claims: ${contract.allowedClaims}`; + }); + return `# Inferred execution profiles\n\nChoose a profile from product stage, mutation authority, risk/evidence needs, and deployment state. Prompt keywords and a request to “be quick” are not routing evidence. A profile narrows or strengthens evidence; it never overrides a specialist’s binding question order, pressure, gates, mutation boundary, or exit behavior.\n\n${rows.join('\n\n')}\n`; +} diff --git a/scripts/gstack2/generate-skill-tree.ts b/scripts/gstack2/generate-skill-tree.ts index db1f73f20..98642a63f 100644 --- a/scripts/gstack2/generate-skill-tree.ts +++ b/scripts/gstack2/generate-skill-tree.ts @@ -5,6 +5,7 @@ import * as path from 'path'; import { BUG_FIX_OVERLAYS, overlaysForSource } from './bug-fix-overlays'; import { renderBrowserProviderContract } from './browser-provider-contract'; import { EXECUTION_RESULT_SCHEMA } from '../../runtime/execution-result.js'; +import { renderExecutionProfiles } from './execution-profiles'; import { contractFor, DISPATCHERS, SOURCE_ASSIGNMENTS } from './assignments'; import { SCENARIOS } from './scenarios'; import { runDeterministicSemanticParity } from './semantic-parity'; @@ -221,7 +222,7 @@ Before any substantive output, print these exact labels in this exact order. Res \`\`\`text Target: Mode: -Depth: +Depth: Mutation: Active modules: Skipped modules: @@ -233,7 +234,7 @@ Web context: 1. Infer the mode from product stage, surface, requested artifact, mutation authorization, evidence needs, and deployment state. Do not route by keyword alone. 2. Refine the public mode to the smallest applicable internal specialist set, then print the required execution header before any substantive output. 3. Read each active module in full from the path shown in the mode/alias tables. Its specialist body, behavioral contract, STOP gates, and appended upstream judgment ports are binding. Read a lazy specialist phase in full only when the workflow reaches its package-local reference. -4. Read \`references/SHARED-JUDGMENT.md\` and \`references/AUTHORITY-POLICY.md\` for every invocation. Read \`references/RUNTIME.md\` before capability-dependent work and \`references/WEB-CONTEXT.md\` before public-web work. +4. Read \`references/EXECUTION-PROFILES.md\`, \`references/SHARED-JUDGMENT.md\`, and \`references/AUTHORITY-POLICY.md\` for every invocation. Infer Depth from structured operating conditions, then obey its mandatory modules, legal skips, artifacts, and claim limits. Read \`references/RUNTIME.md\` before capability-dependent work and \`references/WEB-CONTEXT.md\` before public-web work. 5. If an old asset path is unavailable, use \`references/ASSETS.md\`. If legacy prose invokes another retired skill, resolve it through \`references/COMPATIBILITY.md\` and stay inside these six dispatchers. 6. Preserve report-only versus mutation boundaries. Missing mutation authorization fails closed: do not edit merely because a specialist can fix. Commits, pushes, PRs, merges, deploys, messages, and other external mutations still require affirmative authority from the user. 7. Match the user's language. Keep code identifiers, commands, and source quotations original when translation would reduce accuracy. @@ -567,6 +568,7 @@ function writeSharedContracts(): void { const bootstrap = fs.readFileSync(path.join(ROOT, 'runtime', 'runtime-bootstrap.mjs')); const browserSmoke = fs.readFileSync(path.join(ROOT, 'runtime', 'browser-provider-smoke.mjs')); for (const tree of TREE_NAMES) { + write(path.join(ROOT, 'skills', tree, 'references', 'EXECUTION-PROFILES.md'), `${GENERATED}\n${renderExecutionProfiles()}`); write(path.join(ROOT, 'skills', tree, 'references', 'SHARED-JUDGMENT.md'), sharedJudgmentContract()); write(path.join(ROOT, 'skills', tree, 'references', 'AUTHORITY-POLICY.md'), authorityPolicyContract()); write(path.join(ROOT, 'skills', tree, 'references', 'WEB-CONTEXT.md'), webContextContract()); diff --git a/scripts/gstack2/host-adversarial.ts b/scripts/gstack2/host-adversarial.ts index 7d6244c86..35b87019e 100644 --- a/scripts/gstack2/host-adversarial.ts +++ b/scripts/gstack2/host-adversarial.ts @@ -116,7 +116,7 @@ export interface StructuredHostResult { target: string; skill: PublicSkill; mode: string; - depth: 'quick' | 'standard' | 'deep'; + depth: 'readiness' | 'standard' | 'deep'; mutation: string; active_modules: string[]; skipped_modules: string[]; @@ -236,7 +236,7 @@ export const FINAL_OUTPUT_SCHEMA = { target: { type: 'string' }, skill: { type: 'string', enum: [...PUBLIC_SKILLS] }, mode: { type: 'string' }, - depth: { type: 'string', enum: ['quick', 'standard', 'deep'] }, + depth: { type: 'string', enum: ['readiness', 'standard', 'deep'] }, mutation: { type: 'string' }, active_modules: { type: 'array', items: { type: 'string' } }, skipped_modules: { type: 'array', items: { type: 'string' } }, @@ -625,7 +625,7 @@ export function validateStructuredResult(value: unknown): value is StructuredHos route && typeof route.target === 'string' && PUBLIC_SKILLS.includes(route.skill) && typeof route.mode === 'string' - && ['quick', 'standard', 'deep'].includes(route.depth) + && ['readiness', 'standard', 'deep'].includes(route.depth) && typeof route.mutation === 'string' && isStringArray(route.active_modules) && isStringArray(route.skipped_modules) diff --git a/scripts/gstack2/route.ts b/scripts/gstack2/route.ts index f3cc56e5c..b7ef322c1 100644 --- a/scripts/gstack2/route.ts +++ b/scripts/gstack2/route.ts @@ -1,6 +1,7 @@ import { DISPATCHERS, SOURCE_ASSIGNMENTS, assignmentBySource } from './assignments'; import type { ScenarioFixture, TreeName } from './types'; import { evaluateAuthorityPolicy, type AdversarialAttempt } from './authority-policy'; +import { inferExecutionProfile } from './execution-profiles'; export interface StructuredRoute { tree: TreeName; @@ -154,7 +155,7 @@ export function routeStructured(signals: Record): StructuredRou return { tree, mode, - depth: specialist.defaultDepth, + depth: inferExecutionProfile(signals, specialist.defaultDepth), mutation, active_modules: active, skipped_modules: primary.filter((candidate) => !active.includes(candidate)), diff --git a/scripts/gstack2/types.ts b/scripts/gstack2/types.ts index 5a2ff0bd7..a0132a627 100644 --- a/scripts/gstack2/types.ts +++ b/scripts/gstack2/types.ts @@ -2,6 +2,7 @@ export const GSTACK2_BASE_SHA = 'bb57306d98c97011b0919c6132705a15b1579781'; export const TREE_NAMES = ['plan', 'design', 'qa', 'debug', 'review', 'ship'] as const; export type TreeName = (typeof TREE_NAMES)[number]; +export type ExecutionProfile = 'readiness' | 'standard' | 'deep'; export type ModuleVisibility = 'primary' | 'internal'; @@ -27,7 +28,7 @@ export interface SourceAssignment { mandatory: boolean; replacement: string; summary: string; - defaultDepth: 'quick' | 'standard' | 'deep'; + defaultDepth: ExecutionProfile; defaultMutation: string; webContext: 'none' | 'optional' | 'local-browser' | 'production'; overlays?: number[]; @@ -39,7 +40,7 @@ export interface DispatcherMode { target: string; modules: string[]; inferWhen: string; - depth: 'quick' | 'standard' | 'deep'; + depth: ExecutionProfile; mutation: string; webContext: 'none' | 'optional' | 'local-browser' | 'production'; } @@ -75,7 +76,7 @@ export interface ScenarioFixture { expected: { tree: TreeName; mode: string; - depth: 'quick' | 'standard' | 'deep'; + depth: ExecutionProfile; mutation: string; active_modules: string[]; skipped_modules: string[]; diff --git a/skills/debug/SKILL.md b/skills/debug/SKILL.md index abdbcff64..7c8cf14db 100644 --- a/skills/debug/SKILL.md +++ b/skills/debug/SKILL.md @@ -15,7 +15,7 @@ Before any substantive output, print these exact labels in this exact order. Res ```text Target: Mode: -Depth: +Depth: Mutation: Active modules: Skipped modules: @@ -27,7 +27,7 @@ Web context: 1. Infer the mode from product stage, surface, requested artifact, mutation authorization, evidence needs, and deployment state. Do not route by keyword alone. 2. Refine the public mode to the smallest applicable internal specialist set, then print the required execution header before any substantive output. 3. Read each active module in full from the path shown in the mode/alias tables. Its specialist body, behavioral contract, STOP gates, and appended upstream judgment ports are binding. Read a lazy specialist phase in full only when the workflow reaches its package-local reference. -4. Read `references/SHARED-JUDGMENT.md` and `references/AUTHORITY-POLICY.md` for every invocation. Read `references/RUNTIME.md` before capability-dependent work and `references/WEB-CONTEXT.md` before public-web work. +4. Read `references/EXECUTION-PROFILES.md`, `references/SHARED-JUDGMENT.md`, and `references/AUTHORITY-POLICY.md` for every invocation. Infer Depth from structured operating conditions, then obey its mandatory modules, legal skips, artifacts, and claim limits. Read `references/RUNTIME.md` before capability-dependent work and `references/WEB-CONTEXT.md` before public-web work. 5. If an old asset path is unavailable, use `references/ASSETS.md`. If legacy prose invokes another retired skill, resolve it through `references/COMPATIBILITY.md` and stay inside these six dispatchers. 6. Preserve report-only versus mutation boundaries. Missing mutation authorization fails closed: do not edit merely because a specialist can fix. Commits, pushes, PRs, merges, deploys, messages, and other external mutations still require affirmative authority from the user. 7. Match the user's language. Keep code identifiers, commands, and source quotations original when translation would reduce accuracy. diff --git a/skills/debug/references/EXECUTION-PROFILES.md b/skills/debug/references/EXECUTION-PROFILES.md new file mode 100644 index 000000000..758e32c86 --- /dev/null +++ b/skills/debug/references/EXECUTION-PROFILES.md @@ -0,0 +1,28 @@ + +# Inferred execution profiles + +Choose a profile from product stage, mutation authority, risk/evidence needs, and deployment state. Prompt keywords and a request to “be quick” are not routing evidence. A profile narrows or strengthens evidence; it never overrides a specialist’s binding question order, pressure, gates, mutation boundary, or exit behavior. + +## Smoke/readiness + +- Infer when: A narrow, reversible, pre-deployment or operational readiness decision needs bounded evidence; the requested claim is readiness, not completeness. +- Mandatory modules: Every selected specialist module remains mandatory. Run its entry gates and the smallest source-authorized evidence path that can answer the readiness question. +- Legal skips: Only source-declared conditional work whose condition is demonstrably false, or evidence unavailable after a named attempt. Never skip a STOP gate, approval boundary, reproduction/root-cause gate, or required physical-device/production evidence. +- Artifacts: A readiness record naming the exact scope, probes run, evidence and freshness, failures, skipped work with reasons, and the next standard/deep step. +- Claims: Only ready/not-ready for the named bounded decision. Must say “Readiness profile — not a complete review.” Never claim comprehensive, fully verified, production-safe, or no issues found outside the inspected scope. + +## Standard + +- Infer when: Normal feature or change work has bounded scope and risk and needs the selected specialist’s complete default workflow. +- Mandatory modules: Every selected specialist module and all of its mandatory phases, gates, artifacts, and exit checks. +- Legal skips: Only smart skips explicitly authorized by the specialist and supported by inspected evidence; list each skipped primary module and each skipped conditional phase. +- Artifacts: Every artifact required by the selected specialist, plus evidence provenance, unresolved decisions, and a skip ledger. +- Claims: Complete only for the named specialist scope and evidence layer. Broader product, security, production, or device claims require those modules and evidence. + +## Deep + +- Infer when: Risk, ambiguity, blast radius, cross-system effects, irreversible mutation, security/reliability needs, or production deployment demands stronger evidence. +- Mandatory modules: Every selected specialist module, all mandatory phases and outside-voice/cross-consumer modules selected by the dispatcher, with unchanged STOP and approval gates. +- Legal skips: Only specialist-authorized smart skips proven irrelevant. Missing, stale, malformed, or contradictory evidence is a gap or blocker, never a successful skip. +- Artifacts: All specialist artifacts plus changed-input/unchanged-consumer trace, negative and failure-path evidence, provenance/freshness, rollback or reversibility evidence where applicable, and unresolved-risk ledger. +- Claims: Complete only across the explicitly listed modules and evidence layers. “Confirmed” requires independent supporting evidence; production/device/security claims require matching production/device/security evidence. diff --git a/skills/design/SKILL.md b/skills/design/SKILL.md index 01b06cf8e..c1c8f83a1 100644 --- a/skills/design/SKILL.md +++ b/skills/design/SKILL.md @@ -15,7 +15,7 @@ Before any substantive output, print these exact labels in this exact order. Res ```text Target: Mode: -Depth: +Depth: Mutation: Active modules: Skipped modules: @@ -27,7 +27,7 @@ Web context: 1. Infer the mode from product stage, surface, requested artifact, mutation authorization, evidence needs, and deployment state. Do not route by keyword alone. 2. Refine the public mode to the smallest applicable internal specialist set, then print the required execution header before any substantive output. 3. Read each active module in full from the path shown in the mode/alias tables. Its specialist body, behavioral contract, STOP gates, and appended upstream judgment ports are binding. Read a lazy specialist phase in full only when the workflow reaches its package-local reference. -4. Read `references/SHARED-JUDGMENT.md` and `references/AUTHORITY-POLICY.md` for every invocation. Read `references/RUNTIME.md` before capability-dependent work and `references/WEB-CONTEXT.md` before public-web work. +4. Read `references/EXECUTION-PROFILES.md`, `references/SHARED-JUDGMENT.md`, and `references/AUTHORITY-POLICY.md` for every invocation. Infer Depth from structured operating conditions, then obey its mandatory modules, legal skips, artifacts, and claim limits. Read `references/RUNTIME.md` before capability-dependent work and `references/WEB-CONTEXT.md` before public-web work. 5. If an old asset path is unavailable, use `references/ASSETS.md`. If legacy prose invokes another retired skill, resolve it through `references/COMPATIBILITY.md` and stay inside these six dispatchers. 6. Preserve report-only versus mutation boundaries. Missing mutation authorization fails closed: do not edit merely because a specialist can fix. Commits, pushes, PRs, merges, deploys, messages, and other external mutations still require affirmative authority from the user. 7. Match the user's language. Keep code identifiers, commands, and source quotations original when translation would reduce accuracy. diff --git a/skills/design/references/EXECUTION-PROFILES.md b/skills/design/references/EXECUTION-PROFILES.md new file mode 100644 index 000000000..758e32c86 --- /dev/null +++ b/skills/design/references/EXECUTION-PROFILES.md @@ -0,0 +1,28 @@ + +# Inferred execution profiles + +Choose a profile from product stage, mutation authority, risk/evidence needs, and deployment state. Prompt keywords and a request to “be quick” are not routing evidence. A profile narrows or strengthens evidence; it never overrides a specialist’s binding question order, pressure, gates, mutation boundary, or exit behavior. + +## Smoke/readiness + +- Infer when: A narrow, reversible, pre-deployment or operational readiness decision needs bounded evidence; the requested claim is readiness, not completeness. +- Mandatory modules: Every selected specialist module remains mandatory. Run its entry gates and the smallest source-authorized evidence path that can answer the readiness question. +- Legal skips: Only source-declared conditional work whose condition is demonstrably false, or evidence unavailable after a named attempt. Never skip a STOP gate, approval boundary, reproduction/root-cause gate, or required physical-device/production evidence. +- Artifacts: A readiness record naming the exact scope, probes run, evidence and freshness, failures, skipped work with reasons, and the next standard/deep step. +- Claims: Only ready/not-ready for the named bounded decision. Must say “Readiness profile — not a complete review.” Never claim comprehensive, fully verified, production-safe, or no issues found outside the inspected scope. + +## Standard + +- Infer when: Normal feature or change work has bounded scope and risk and needs the selected specialist’s complete default workflow. +- Mandatory modules: Every selected specialist module and all of its mandatory phases, gates, artifacts, and exit checks. +- Legal skips: Only smart skips explicitly authorized by the specialist and supported by inspected evidence; list each skipped primary module and each skipped conditional phase. +- Artifacts: Every artifact required by the selected specialist, plus evidence provenance, unresolved decisions, and a skip ledger. +- Claims: Complete only for the named specialist scope and evidence layer. Broader product, security, production, or device claims require those modules and evidence. + +## Deep + +- Infer when: Risk, ambiguity, blast radius, cross-system effects, irreversible mutation, security/reliability needs, or production deployment demands stronger evidence. +- Mandatory modules: Every selected specialist module, all mandatory phases and outside-voice/cross-consumer modules selected by the dispatcher, with unchanged STOP and approval gates. +- Legal skips: Only specialist-authorized smart skips proven irrelevant. Missing, stale, malformed, or contradictory evidence is a gap or blocker, never a successful skip. +- Artifacts: All specialist artifacts plus changed-input/unchanged-consumer trace, negative and failure-path evidence, provenance/freshness, rollback or reversibility evidence where applicable, and unresolved-risk ledger. +- Claims: Complete only across the explicitly listed modules and evidence layers. “Confirmed” requires independent supporting evidence; production/device/security claims require matching production/device/security evidence. diff --git a/skills/plan/SKILL.md b/skills/plan/SKILL.md index f9913003c..8c06285b8 100644 --- a/skills/plan/SKILL.md +++ b/skills/plan/SKILL.md @@ -15,7 +15,7 @@ Before any substantive output, print these exact labels in this exact order. Res ```text Target: Mode: -Depth: +Depth: Mutation: Active modules: Skipped modules: @@ -27,7 +27,7 @@ Web context: 1. Infer the mode from product stage, surface, requested artifact, mutation authorization, evidence needs, and deployment state. Do not route by keyword alone. 2. Refine the public mode to the smallest applicable internal specialist set, then print the required execution header before any substantive output. 3. Read each active module in full from the path shown in the mode/alias tables. Its specialist body, behavioral contract, STOP gates, and appended upstream judgment ports are binding. Read a lazy specialist phase in full only when the workflow reaches its package-local reference. -4. Read `references/SHARED-JUDGMENT.md` and `references/AUTHORITY-POLICY.md` for every invocation. Read `references/RUNTIME.md` before capability-dependent work and `references/WEB-CONTEXT.md` before public-web work. +4. Read `references/EXECUTION-PROFILES.md`, `references/SHARED-JUDGMENT.md`, and `references/AUTHORITY-POLICY.md` for every invocation. Infer Depth from structured operating conditions, then obey its mandatory modules, legal skips, artifacts, and claim limits. Read `references/RUNTIME.md` before capability-dependent work and `references/WEB-CONTEXT.md` before public-web work. 5. If an old asset path is unavailable, use `references/ASSETS.md`. If legacy prose invokes another retired skill, resolve it through `references/COMPATIBILITY.md` and stay inside these six dispatchers. 6. Preserve report-only versus mutation boundaries. Missing mutation authorization fails closed: do not edit merely because a specialist can fix. Commits, pushes, PRs, merges, deploys, messages, and other external mutations still require affirmative authority from the user. 7. Match the user's language. Keep code identifiers, commands, and source quotations original when translation would reduce accuracy. diff --git a/skills/plan/references/EXECUTION-PROFILES.md b/skills/plan/references/EXECUTION-PROFILES.md new file mode 100644 index 000000000..758e32c86 --- /dev/null +++ b/skills/plan/references/EXECUTION-PROFILES.md @@ -0,0 +1,28 @@ + +# Inferred execution profiles + +Choose a profile from product stage, mutation authority, risk/evidence needs, and deployment state. Prompt keywords and a request to “be quick” are not routing evidence. A profile narrows or strengthens evidence; it never overrides a specialist’s binding question order, pressure, gates, mutation boundary, or exit behavior. + +## Smoke/readiness + +- Infer when: A narrow, reversible, pre-deployment or operational readiness decision needs bounded evidence; the requested claim is readiness, not completeness. +- Mandatory modules: Every selected specialist module remains mandatory. Run its entry gates and the smallest source-authorized evidence path that can answer the readiness question. +- Legal skips: Only source-declared conditional work whose condition is demonstrably false, or evidence unavailable after a named attempt. Never skip a STOP gate, approval boundary, reproduction/root-cause gate, or required physical-device/production evidence. +- Artifacts: A readiness record naming the exact scope, probes run, evidence and freshness, failures, skipped work with reasons, and the next standard/deep step. +- Claims: Only ready/not-ready for the named bounded decision. Must say “Readiness profile — not a complete review.” Never claim comprehensive, fully verified, production-safe, or no issues found outside the inspected scope. + +## Standard + +- Infer when: Normal feature or change work has bounded scope and risk and needs the selected specialist’s complete default workflow. +- Mandatory modules: Every selected specialist module and all of its mandatory phases, gates, artifacts, and exit checks. +- Legal skips: Only smart skips explicitly authorized by the specialist and supported by inspected evidence; list each skipped primary module and each skipped conditional phase. +- Artifacts: Every artifact required by the selected specialist, plus evidence provenance, unresolved decisions, and a skip ledger. +- Claims: Complete only for the named specialist scope and evidence layer. Broader product, security, production, or device claims require those modules and evidence. + +## Deep + +- Infer when: Risk, ambiguity, blast radius, cross-system effects, irreversible mutation, security/reliability needs, or production deployment demands stronger evidence. +- Mandatory modules: Every selected specialist module, all mandatory phases and outside-voice/cross-consumer modules selected by the dispatcher, with unchanged STOP and approval gates. +- Legal skips: Only specialist-authorized smart skips proven irrelevant. Missing, stale, malformed, or contradictory evidence is a gap or blocker, never a successful skip. +- Artifacts: All specialist artifacts plus changed-input/unchanged-consumer trace, negative and failure-path evidence, provenance/freshness, rollback or reversibility evidence where applicable, and unresolved-risk ledger. +- Claims: Complete only across the explicitly listed modules and evidence layers. “Confirmed” requires independent supporting evidence; production/device/security claims require matching production/device/security evidence. diff --git a/skills/qa/SKILL.md b/skills/qa/SKILL.md index a3dd07f4a..a51f97746 100644 --- a/skills/qa/SKILL.md +++ b/skills/qa/SKILL.md @@ -15,7 +15,7 @@ Before any substantive output, print these exact labels in this exact order. Res ```text Target: Mode: -Depth: +Depth: Mutation: Active modules: Skipped modules: @@ -27,7 +27,7 @@ Web context: 1. Infer the mode from product stage, surface, requested artifact, mutation authorization, evidence needs, and deployment state. Do not route by keyword alone. 2. Refine the public mode to the smallest applicable internal specialist set, then print the required execution header before any substantive output. 3. Read each active module in full from the path shown in the mode/alias tables. Its specialist body, behavioral contract, STOP gates, and appended upstream judgment ports are binding. Read a lazy specialist phase in full only when the workflow reaches its package-local reference. -4. Read `references/SHARED-JUDGMENT.md` and `references/AUTHORITY-POLICY.md` for every invocation. Read `references/RUNTIME.md` before capability-dependent work and `references/WEB-CONTEXT.md` before public-web work. +4. Read `references/EXECUTION-PROFILES.md`, `references/SHARED-JUDGMENT.md`, and `references/AUTHORITY-POLICY.md` for every invocation. Infer Depth from structured operating conditions, then obey its mandatory modules, legal skips, artifacts, and claim limits. Read `references/RUNTIME.md` before capability-dependent work and `references/WEB-CONTEXT.md` before public-web work. 5. If an old asset path is unavailable, use `references/ASSETS.md`. If legacy prose invokes another retired skill, resolve it through `references/COMPATIBILITY.md` and stay inside these six dispatchers. 6. Preserve report-only versus mutation boundaries. Missing mutation authorization fails closed: do not edit merely because a specialist can fix. Commits, pushes, PRs, merges, deploys, messages, and other external mutations still require affirmative authority from the user. 7. Match the user's language. Keep code identifiers, commands, and source quotations original when translation would reduce accuracy. diff --git a/skills/qa/references/EXECUTION-PROFILES.md b/skills/qa/references/EXECUTION-PROFILES.md new file mode 100644 index 000000000..758e32c86 --- /dev/null +++ b/skills/qa/references/EXECUTION-PROFILES.md @@ -0,0 +1,28 @@ + +# Inferred execution profiles + +Choose a profile from product stage, mutation authority, risk/evidence needs, and deployment state. Prompt keywords and a request to “be quick” are not routing evidence. A profile narrows or strengthens evidence; it never overrides a specialist’s binding question order, pressure, gates, mutation boundary, or exit behavior. + +## Smoke/readiness + +- Infer when: A narrow, reversible, pre-deployment or operational readiness decision needs bounded evidence; the requested claim is readiness, not completeness. +- Mandatory modules: Every selected specialist module remains mandatory. Run its entry gates and the smallest source-authorized evidence path that can answer the readiness question. +- Legal skips: Only source-declared conditional work whose condition is demonstrably false, or evidence unavailable after a named attempt. Never skip a STOP gate, approval boundary, reproduction/root-cause gate, or required physical-device/production evidence. +- Artifacts: A readiness record naming the exact scope, probes run, evidence and freshness, failures, skipped work with reasons, and the next standard/deep step. +- Claims: Only ready/not-ready for the named bounded decision. Must say “Readiness profile — not a complete review.” Never claim comprehensive, fully verified, production-safe, or no issues found outside the inspected scope. + +## Standard + +- Infer when: Normal feature or change work has bounded scope and risk and needs the selected specialist’s complete default workflow. +- Mandatory modules: Every selected specialist module and all of its mandatory phases, gates, artifacts, and exit checks. +- Legal skips: Only smart skips explicitly authorized by the specialist and supported by inspected evidence; list each skipped primary module and each skipped conditional phase. +- Artifacts: Every artifact required by the selected specialist, plus evidence provenance, unresolved decisions, and a skip ledger. +- Claims: Complete only for the named specialist scope and evidence layer. Broader product, security, production, or device claims require those modules and evidence. + +## Deep + +- Infer when: Risk, ambiguity, blast radius, cross-system effects, irreversible mutation, security/reliability needs, or production deployment demands stronger evidence. +- Mandatory modules: Every selected specialist module, all mandatory phases and outside-voice/cross-consumer modules selected by the dispatcher, with unchanged STOP and approval gates. +- Legal skips: Only specialist-authorized smart skips proven irrelevant. Missing, stale, malformed, or contradictory evidence is a gap or blocker, never a successful skip. +- Artifacts: All specialist artifacts plus changed-input/unchanged-consumer trace, negative and failure-path evidence, provenance/freshness, rollback or reversibility evidence where applicable, and unresolved-risk ledger. +- Claims: Complete only across the explicitly listed modules and evidence layers. “Confirmed” requires independent supporting evidence; production/device/security claims require matching production/device/security evidence. diff --git a/skills/review/SKILL.md b/skills/review/SKILL.md index fce9c7c3b..1f511d469 100644 --- a/skills/review/SKILL.md +++ b/skills/review/SKILL.md @@ -15,7 +15,7 @@ Before any substantive output, print these exact labels in this exact order. Res ```text Target: Mode: -Depth: +Depth: Mutation: Active modules: Skipped modules: @@ -27,7 +27,7 @@ Web context: 1. Infer the mode from product stage, surface, requested artifact, mutation authorization, evidence needs, and deployment state. Do not route by keyword alone. 2. Refine the public mode to the smallest applicable internal specialist set, then print the required execution header before any substantive output. 3. Read each active module in full from the path shown in the mode/alias tables. Its specialist body, behavioral contract, STOP gates, and appended upstream judgment ports are binding. Read a lazy specialist phase in full only when the workflow reaches its package-local reference. -4. Read `references/SHARED-JUDGMENT.md` and `references/AUTHORITY-POLICY.md` for every invocation. Read `references/RUNTIME.md` before capability-dependent work and `references/WEB-CONTEXT.md` before public-web work. +4. Read `references/EXECUTION-PROFILES.md`, `references/SHARED-JUDGMENT.md`, and `references/AUTHORITY-POLICY.md` for every invocation. Infer Depth from structured operating conditions, then obey its mandatory modules, legal skips, artifacts, and claim limits. Read `references/RUNTIME.md` before capability-dependent work and `references/WEB-CONTEXT.md` before public-web work. 5. If an old asset path is unavailable, use `references/ASSETS.md`. If legacy prose invokes another retired skill, resolve it through `references/COMPATIBILITY.md` and stay inside these six dispatchers. 6. Preserve report-only versus mutation boundaries. Missing mutation authorization fails closed: do not edit merely because a specialist can fix. Commits, pushes, PRs, merges, deploys, messages, and other external mutations still require affirmative authority from the user. 7. Match the user's language. Keep code identifiers, commands, and source quotations original when translation would reduce accuracy. diff --git a/skills/review/references/EXECUTION-PROFILES.md b/skills/review/references/EXECUTION-PROFILES.md new file mode 100644 index 000000000..758e32c86 --- /dev/null +++ b/skills/review/references/EXECUTION-PROFILES.md @@ -0,0 +1,28 @@ + +# Inferred execution profiles + +Choose a profile from product stage, mutation authority, risk/evidence needs, and deployment state. Prompt keywords and a request to “be quick” are not routing evidence. A profile narrows or strengthens evidence; it never overrides a specialist’s binding question order, pressure, gates, mutation boundary, or exit behavior. + +## Smoke/readiness + +- Infer when: A narrow, reversible, pre-deployment or operational readiness decision needs bounded evidence; the requested claim is readiness, not completeness. +- Mandatory modules: Every selected specialist module remains mandatory. Run its entry gates and the smallest source-authorized evidence path that can answer the readiness question. +- Legal skips: Only source-declared conditional work whose condition is demonstrably false, or evidence unavailable after a named attempt. Never skip a STOP gate, approval boundary, reproduction/root-cause gate, or required physical-device/production evidence. +- Artifacts: A readiness record naming the exact scope, probes run, evidence and freshness, failures, skipped work with reasons, and the next standard/deep step. +- Claims: Only ready/not-ready for the named bounded decision. Must say “Readiness profile — not a complete review.” Never claim comprehensive, fully verified, production-safe, or no issues found outside the inspected scope. + +## Standard + +- Infer when: Normal feature or change work has bounded scope and risk and needs the selected specialist’s complete default workflow. +- Mandatory modules: Every selected specialist module and all of its mandatory phases, gates, artifacts, and exit checks. +- Legal skips: Only smart skips explicitly authorized by the specialist and supported by inspected evidence; list each skipped primary module and each skipped conditional phase. +- Artifacts: Every artifact required by the selected specialist, plus evidence provenance, unresolved decisions, and a skip ledger. +- Claims: Complete only for the named specialist scope and evidence layer. Broader product, security, production, or device claims require those modules and evidence. + +## Deep + +- Infer when: Risk, ambiguity, blast radius, cross-system effects, irreversible mutation, security/reliability needs, or production deployment demands stronger evidence. +- Mandatory modules: Every selected specialist module, all mandatory phases and outside-voice/cross-consumer modules selected by the dispatcher, with unchanged STOP and approval gates. +- Legal skips: Only specialist-authorized smart skips proven irrelevant. Missing, stale, malformed, or contradictory evidence is a gap or blocker, never a successful skip. +- Artifacts: All specialist artifacts plus changed-input/unchanged-consumer trace, negative and failure-path evidence, provenance/freshness, rollback or reversibility evidence where applicable, and unresolved-risk ledger. +- Claims: Complete only across the explicitly listed modules and evidence layers. “Confirmed” requires independent supporting evidence; production/device/security claims require matching production/device/security evidence. diff --git a/skills/ship/SKILL.md b/skills/ship/SKILL.md index 152cdc211..599c120f0 100644 --- a/skills/ship/SKILL.md +++ b/skills/ship/SKILL.md @@ -15,7 +15,7 @@ Before any substantive output, print these exact labels in this exact order. Res ```text Target: Mode: -Depth: +Depth: Mutation: Active modules: Skipped modules: @@ -27,7 +27,7 @@ Web context: 1. Infer the mode from product stage, surface, requested artifact, mutation authorization, evidence needs, and deployment state. Do not route by keyword alone. 2. Refine the public mode to the smallest applicable internal specialist set, then print the required execution header before any substantive output. 3. Read each active module in full from the path shown in the mode/alias tables. Its specialist body, behavioral contract, STOP gates, and appended upstream judgment ports are binding. Read a lazy specialist phase in full only when the workflow reaches its package-local reference. -4. Read `references/SHARED-JUDGMENT.md` and `references/AUTHORITY-POLICY.md` for every invocation. Read `references/RUNTIME.md` before capability-dependent work and `references/WEB-CONTEXT.md` before public-web work. +4. Read `references/EXECUTION-PROFILES.md`, `references/SHARED-JUDGMENT.md`, and `references/AUTHORITY-POLICY.md` for every invocation. Infer Depth from structured operating conditions, then obey its mandatory modules, legal skips, artifacts, and claim limits. Read `references/RUNTIME.md` before capability-dependent work and `references/WEB-CONTEXT.md` before public-web work. 5. If an old asset path is unavailable, use `references/ASSETS.md`. If legacy prose invokes another retired skill, resolve it through `references/COMPATIBILITY.md` and stay inside these six dispatchers. 6. Preserve report-only versus mutation boundaries. Missing mutation authorization fails closed: do not edit merely because a specialist can fix. Commits, pushes, PRs, merges, deploys, messages, and other external mutations still require affirmative authority from the user. 7. Match the user's language. Keep code identifiers, commands, and source quotations original when translation would reduce accuracy. diff --git a/skills/ship/references/EXECUTION-PROFILES.md b/skills/ship/references/EXECUTION-PROFILES.md new file mode 100644 index 000000000..758e32c86 --- /dev/null +++ b/skills/ship/references/EXECUTION-PROFILES.md @@ -0,0 +1,28 @@ + +# Inferred execution profiles + +Choose a profile from product stage, mutation authority, risk/evidence needs, and deployment state. Prompt keywords and a request to “be quick” are not routing evidence. A profile narrows or strengthens evidence; it never overrides a specialist’s binding question order, pressure, gates, mutation boundary, or exit behavior. + +## Smoke/readiness + +- Infer when: A narrow, reversible, pre-deployment or operational readiness decision needs bounded evidence; the requested claim is readiness, not completeness. +- Mandatory modules: Every selected specialist module remains mandatory. Run its entry gates and the smallest source-authorized evidence path that can answer the readiness question. +- Legal skips: Only source-declared conditional work whose condition is demonstrably false, or evidence unavailable after a named attempt. Never skip a STOP gate, approval boundary, reproduction/root-cause gate, or required physical-device/production evidence. +- Artifacts: A readiness record naming the exact scope, probes run, evidence and freshness, failures, skipped work with reasons, and the next standard/deep step. +- Claims: Only ready/not-ready for the named bounded decision. Must say “Readiness profile — not a complete review.” Never claim comprehensive, fully verified, production-safe, or no issues found outside the inspected scope. + +## Standard + +- Infer when: Normal feature or change work has bounded scope and risk and needs the selected specialist’s complete default workflow. +- Mandatory modules: Every selected specialist module and all of its mandatory phases, gates, artifacts, and exit checks. +- Legal skips: Only smart skips explicitly authorized by the specialist and supported by inspected evidence; list each skipped primary module and each skipped conditional phase. +- Artifacts: Every artifact required by the selected specialist, plus evidence provenance, unresolved decisions, and a skip ledger. +- Claims: Complete only for the named specialist scope and evidence layer. Broader product, security, production, or device claims require those modules and evidence. + +## Deep + +- Infer when: Risk, ambiguity, blast radius, cross-system effects, irreversible mutation, security/reliability needs, or production deployment demands stronger evidence. +- Mandatory modules: Every selected specialist module, all mandatory phases and outside-voice/cross-consumer modules selected by the dispatcher, with unchanged STOP and approval gates. +- Legal skips: Only specialist-authorized smart skips proven irrelevant. Missing, stale, malformed, or contradictory evidence is a gap or blocker, never a successful skip. +- Artifacts: All specialist artifacts plus changed-input/unchanged-consumer trace, negative and failure-path evidence, provenance/freshness, rollback or reversibility evidence where applicable, and unresolved-risk ledger. +- Claims: Complete only across the explicitly listed modules and evidence layers. “Confirmed” requires independent supporting evidence; production/device/security claims require matching production/device/security evidence. diff --git a/test/gstack2-skill-ux.test.ts b/test/gstack2-skill-ux.test.ts index 1a7dcdaf3..af647b343 100644 --- a/test/gstack2-skill-ux.test.ts +++ b/test/gstack2-skill-ux.test.ts @@ -45,6 +45,21 @@ describe('GStack 2 canonical skill UX', () => { } }); + test('packages one binding inferred execution-profile contract in every dispatcher', () => { + for (const tree of TREE_NAMES) { + const dispatcher = fs.readFileSync(path.join(ROOT, 'skills', tree, 'SKILL.md'), 'utf8'); + const profiles = fs.readFileSync(path.join(ROOT, 'skills', tree, 'references', 'EXECUTION-PROFILES.md'), 'utf8'); + expect(dispatcher, tree).toContain('Depth: '); + expect(dispatcher, tree).toContain('Read `references/EXECUTION-PROFILES.md`'); + expect(profiles, tree).toContain('## Smoke/readiness'); + expect(profiles, tree).toContain('Readiness profile — not a complete review.'); + expect(profiles, tree).toContain('Every selected specialist module remains mandatory.'); + expect(profiles, tree).toContain('## Standard'); + expect(profiles, tree).toContain('## Deep'); + expect(profiles, tree).toContain('never overrides a specialist’s binding question order'); + } + }); + test('resolves retired user-facing recommendations without rewriting package paths', () => { for (const assignment of SOURCE_ASSIGNMENTS) { const body = fs.readFileSync(ownerModule(assignment.source), 'utf8'); diff --git a/test/gstack2-skills-routing.test.ts b/test/gstack2-skills-routing.test.ts index bda2f4ce1..44bb85abb 100644 --- a/test/gstack2-skills-routing.test.ts +++ b/test/gstack2-skills-routing.test.ts @@ -24,6 +24,41 @@ describe('GStack 2 structured dispatch', () => { expect(routeStructured({ ...scenario.signals })).toEqual(original); }); + test('infers readiness only from bounded structured operating conditions', () => { + const readiness = routeStructured({ + surface: 'web', + implementation_exists: true, + evidence_need: 'readiness', + scope: 'narrow', + deployment_state: 'pre-deployment', + }); + expect(readiness.depth).toBe('readiness'); + expect(readiness.active_modules).toEqual(['qa-only']); + + expect(routeStructured({ + surface: 'web', + implementation_exists: true, + evidence_need: 'readiness', + scope: 'broad', + }).depth).not.toBe('readiness'); + }); + + test('risk and deployment evidence promote work to deep', () => { + expect(routeStructured({ surface: 'web', implementation_exists: true, risk: 'high' }).depth) + .toBe('deep'); + expect(routeStructured({ surface: 'web', implementation_exists: true, deployment_state: 'production' }).depth) + .toBe('deep'); + expect(routeStructured({ surface: 'web', implementation_exists: true, mutation_scope: 'consequential' }).depth) + .toBe('deep'); + expect(routeStructured({ + surface: 'web', + implementation_exists: true, + evidence_need: 'readiness', + scope: 'narrow', + irreversible: true, + }).depth).toBe('deep'); + }); + test('explicit mutation denials override otherwise mutating modes', () => { const review = routeStructured({ audit_focus: 'broad', mutation_authorized: false }); expect(review.mode).toBe('Normal');