From 4973f96d98a1d0a078eec522681d8060e6f80dd2 Mon Sep 17 00:00:00 2001 From: garrytan Date: Tue, 29 Sep 2026 15:29:46 +0000 Subject: [PATCH] test: move the full office-hours workflow to marathon; add a periodic design-draft checkpoint MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The full startup workflow runs 1–3 real spec-review rounds (~280s each) and hit its 1200s capture in run 36385945043 at finalize. Review depth is the product's loop, so the case cannot fit a blocking lane without cutting rounds. It is now marathon tier with every assertion unchanged. skill-e2e-office-hours-design-draft.test.ts (periodic) runs the same fixed interview only through the Write that creates the design (269s in that run) and applies the full validator's design-draft checks, the required section reads and the launch/foreign-skill-read guards. validateOfficeHoursDesignDraft is extracted from validateOfficeHoursCompletion, which still applies it. Selection: office-hours-design-draft is registered periodic; the marathon-only file is already excluded from the gate and periodic plans by the B5 planner rule. Tier-alignment regexes and the valid-tier check accept 'marathon'. A type-only cast in plan-scope-selection.test.ts removes a diagnostic whose union print order made the ratchet identity unstable; baseline tightened. --- scripts/typecheck-test-baseline.json | 2 - test/ceo-split-collection.test.ts | 17 +++-- test/e2e-tier-alignment.test.ts | 4 +- test/helpers/office-hours-completion.ts | 71 ++++++++++++------- test/helpers/touchfiles-data.ts | 4 +- test/office-hours-budget.test.ts | 36 +++++++--- test/office-hours-completion.test.ts | 18 ++++- test/plan-scope-selection.test.ts | 2 +- test/qa-supervision-selection.test.ts | 2 +- ...kill-e2e-office-hours-design-draft.test.ts | 68 ++++++++++++++++++ ...l-e2e-office-hours-section-loading.test.ts | 11 ++- test/touchfiles.test.ts | 2 +- 12 files changed, 180 insertions(+), 57 deletions(-) create mode 100644 test/skill-e2e-office-hours-design-draft.test.ts diff --git a/scripts/typecheck-test-baseline.json b/scripts/typecheck-test-baseline.json index bb83f5d76..4f4d91add 100644 --- a/scripts/typecheck-test-baseline.json +++ b/scripts/typecheck-test-baseline.json @@ -213,7 +213,6 @@ "test/helpers/ceo-hold-posture-review.ts\tTS2339\tProperty 'toolUseId' does not exist on type 'AskUserQuestionFingerprint'.": 2, "test/helpers/ceo-mode-option.ts\tTS2345\tArgument of type 'string' is not assignable to parameter of type '\"defer\" | \"include\" | \"pause\" | \"skip\" | null'.": 1, "test/helpers/ceo-split-question-policy.ts\tTS2345\tArgument of type 'NativePlanQuestion' is not assignable to parameter of type 'NativeQuestion'. Types of property 'multiSelect' are incompatible. Type 'boolean | undefined' is not assignable to type 'boolean'. Type 'undefined' is not assignable to type 'boolean'.": 3, - "test/helpers/ceo-split-question-policy.ts\tTS2345\tArgument of type 'string' is not assignable to parameter of type '\"cut\" | \"defer\" | \"hold\" | \"include\" | null'.": 1, "test/helpers/claude-pty-runner.unit.test.ts\tTS2322\tType '{ [x: string]: string; }' is not assignable to type '{ \"D1 \\u2014 Add gstack skill routing rules to CLAUDE.md? \": string; \"D2 \\u2014 Should gstack search learnings from your other projects on this machine? \"?: undefined; ... 7 more ...; \"D10 \\u2014 TODO: Add p99 latency metric for IDP calls before/after...'. Property '\"D10 — TODO: Add p99 latency metric for IDP calls before/after Promise.all parallelization. Add to TODOS.md? \"' is missing in type '{ [x: string]: string; }' but required in type '{ \"D1 \\u2014 Add gstack skill routing rules to CLAUDE.md? \"?: undefined; \"D2 \\u2014 Should gstack search learnings from your other projects on this machine? \"?: undefined; ... 7 more ...; \"D10 \\u2014 TODO: Add p99 latency metric for IDP calls before/a...'.": 3, "test/helpers/claude-pty-runner.unit.test.ts\tTS2322\tType '{ [x: string]: string; }' is not assignable to type '{ \"D1 \\u2014 The plan's scope (12 files, 4 new classes) triggers the complexity smell check. Proceed as-is or reduce scope first? \": string; ... 7 more ...; \"D9 \\u2014 TODOS: the plan has no mention of IDP circuit breaker or timeout per call. With 5 calls now running in parallel...'. Property '\"D9 — TODOS: the plan has no mention of IDP circuit breaker or timeout per call. With 5 calls now running in parallel (D8 decision), an IDP outage generates 5 concurrent timeouts per request. \"' is missing in type '{ [x: string]: string; }' but required in type '{ \"D1 \\u2014 The plan's scope (12 files, 4 new classes) triggers the complexity smell check. Proceed as-is or reduce scope first? \"?: undefined; ... 7 more ...; \"D9 \\u2014 TODOS: the plan has no mention of IDP circuit breaker or timeout per call. With 5 calls now running in para...'.": 1, "test/helpers/claude-pty-runner.unit.test.ts\tTS2322\tType '{}' is not assignable to type '{ \"D1 \\u2014 Add gstack skill routing rules to CLAUDE.md? \": string; \"D2 \\u2014 Should gstack search learnings from your other projects on this machine? \"?: undefined; ... 7 more ...; \"D10 \\u2014 TODO: Add p99 latency metric for IDP calls before/after...'.": 1, @@ -416,7 +415,6 @@ "test/plan-review-decisions.test.ts\tTS7006\tParameter 'row' implicitly has an 'any' type.": 1, "test/plan-scope-selection.test.ts\tTS2339\tProperty 'args' does not exist on type '{ skill: string; args: string; } | { skill: string; }'. Property 'args' does not exist on type '{ skill: string; }'.": 3, "test/plan-scope-selection.test.ts\tTS2339\tProperty 'args' does not exist on type '{ skill: string; } | { skill: string; args: string; }'. Property 'args' does not exist on type '{ skill: string; }'.": 4, - "test/plan-scope-selection.test.ts\tTS2345\tArgument of type '{ isError?: undefined; kind: string; sessionId: string; timestamp: string; name: string; toolUseId: string; input: { skill: string; }; } | { name?: undefined; input?: undefined; kind: string; sessionId: string; timestamp: string; toolUseId: string; isError: boolean; } | { ...; } | { ...; } | { ...; } | { ...; }' is not assignable to parameter of type '{ isError?: undefined; kind: string; sessionId: string; timestamp: string; name: string; toolUseId: string; input: { skill: string; args: string; }; } | { name?: undefined; input?: undefined; kind: string; sessionId: string; timestamp: string; toolUseId: string; isError: boolean; } | { ...; }'. Type '{ isError?: undefined; kind: string; sessionId: string; timestamp: string; name: string; toolUseId: string; input: { skill: string; }; }' is not assignable to type '{ isError?: undefined; kind: string; sessionId: string; timestamp: string; name: string; toolUseId: string; input: { skill: string; args: string; }; } | { name?: undefined; input?: undefined; kind: string; sessionId: string; timestamp: string; toolUseId: string; isError: boolean; } | { ...; }'. Type '{ isError?: undefined; kind: string; sessionId: string; timestamp: string; name: string; toolUseId: string; input: { skill: string; }; }' is not assignable to type '{ isError?: undefined; kind: string; sessionId: string; timestamp: string; name: string; toolUseId: string; input: { skill: string; args: string; }; } | { isError?: undefined; input?: undefined; kind: string; sessionId: string; timestamp: string; name: string; toolUseId: string; }'. Type '{ isError?: undefined; kind: string; sessionId: string; timestamp: string; name: string; toolUseId: string; input: { skill: string; }; }' is not assignable to type '{ isError?: undefined; input?: undefined; kind: string; sessionId: string; timestamp: string; name: string; toolUseId: string; }'. Types of property '\"input\"' are incompatible. Type '{ skill: string; }' is not assignable to type 'undefined'.": 1, "test/plan-seed-submission.test.ts\tTS2322\tType '{ GSTACK_PLAN_MODE?: undefined; CLAUDE_CONFIG_DIR: string; SEED_CASE: string; } | { GSTACK_PLAN_MODE: string; CLAUDE_CONFIG_DIR: string; SEED_CASE: string; } | { GSTACK_PLAN_MODE?: undefined; CLAUDE_CONFIG_DIR: string; SEED_CASE: string; } | { ...; }' is not assignable to type 'Record | undefined'. Type '{ GSTACK_PLAN_MODE?: undefined; CLAUDE_CONFIG_DIR: string; SEED_CASE: string; }' is not assignable to type 'Record'. Property 'GSTACK_PLAN_MODE' is incompatible with index signature. Type 'undefined' is not assignable to type 'string'.": 1, "test/plan-seed-submission.test.ts\tTS2322\tType '{ TERM: string; CLAUDE_CONFIG_DIR: string; SEED_CASE: string; } | { CI: string; TERM?: undefined; COLORTERM?: undefined; FORCE_COLOR?: undefined; NO_COLOR?: undefined; CLAUDE_CONFIG_DIR: string; SEED_CASE: string; } | { ...; } | { ...; } | { ...; }' is not assignable to type 'Record | undefined'. Type '{ CI: string; TERM?: undefined; COLORTERM?: undefined; FORCE_COLOR?: undefined; NO_COLOR?: undefined; CLAUDE_CONFIG_DIR: string; SEED_CASE: string; }' is not assignable to type 'Record'. Property 'TERM' is incompatible with index signature. Type 'undefined' is not assignable to type 'string'.": 1, "test/plan-seed-submission.test.ts\tTS2345\tArgument of type '(text: string, reviver?: ((this: any, key: string, value: any) => any) | undefined) => any' is not assignable to parameter of type '(value: string, index: number, array: string[]) => any'. Types of parameters 'reviver' and 'index' are incompatible. Type 'number' is not assignable to type '(this: any, key: string, value: any) => any'.": 4, diff --git a/test/ceo-split-collection.test.ts b/test/ceo-split-collection.test.ts index 70c2d79e1..17a0da978 100644 --- a/test/ceo-split-collection.test.ts +++ b/test/ceo-split-collection.test.ts @@ -2,7 +2,8 @@ import { expect, test } from 'bun:test'; import * as fs from 'node:fs'; import * as os from 'node:os'; import * as path from 'node:path'; -import { nativePlanCallFingerprint } from './helpers/claude-pty-runner'; +import { nativePlanCallFingerprint, type AskUserQuestionFingerprint } from './helpers/claude-pty-runner'; +import type { NativeQuestion } from './helpers/plan-skill-questions'; import type { NativePlanQuestionCall, PlanCountTranscript } from './helpers/plan-count-transcript'; import { ceoSplitCandidate, ceoSplitDecisionFingerprints, isCeoSplitCandidateCall, isCeoSplitCollectionComplete } from './helpers/ceo-split-question-policy'; import captured from './fixtures/ceo-split-collection-0bcd.json'; @@ -54,26 +55,28 @@ test.each([0, 1, 2, 3, 4, 5, 6])('the exact original %i-call prefix waits for th // Run 36385945043: the skill cited ledger row IDs ("D2.1 — R-E1: …") and offered // a fourth "Hold, discuss first" option. No candidate was recognized, so collection // never stopped and the attempt ran the whole review (1302s) after the E5 ACK. -function rowIdCapture() { - const calls = structuredClone(rowIds.calls) as NativePlanQuestionCall[]; +function rowIdCapture(): { transcript: PlanCountTranscript; fingerprints: AskUserQuestionFingerprint[] } { + const calls = structuredClone(rowIds.calls) as unknown as NativePlanQuestionCall[]; const transcript: PlanCountTranscript = { status: 'ready', calls, assistantMessages: [] }; const fingerprints = rowIds.fingerprints.map((fp, index) => ({ ...structuredClone(fp), nativeCall: calls[index]! })); return { transcript, fingerprints }; } +const rowIdAccepts = (state: ReturnType) => isCeoSplitCollectionComplete(state.transcript, state.fingerprints); test('ledger row-ID candidate questions from run 36385945043 finish collection at the E5 ACK', () => { const state = rowIdCapture(); expect(rowIds.provenance.originalOutcome).toBe('completion_summary'); expect(rowIds.provenance.originalReviewCount).toBe(0); expect(state.transcript.calls.at(-1)!.answeredAt).toBe(rowIds.provenance.completeAt); - expect(state.transcript.calls.map(call => ceoSplitCandidate(call.questions[0]!))).toEqual([null, 'E1', 'E2', 'E3', 'E4', 'E5']); + expect(state.transcript.calls.map(call => ceoSplitCandidate(call.questions[0] as NativeQuestion))) + .toEqual([null, 'E1', 'E2', 'E3', 'E4', 'E5']); expect(state.fingerprints.map(isCeoSplitCandidateCall)).toEqual([false, true, true, true, true, true]); for (let length = 0; length < 6; length++) { const prefix = rowIdCapture(); prefix.transcript.calls.length = length; prefix.fingerprints.length = length; - expect(accepts(prefix)).toBe(false); + expect(rowIdAccepts(prefix)).toBe(false); } - expect(accepts(state)).toBe(true); + expect(rowIdAccepts(state)).toBe(true); }); test.each(['foreign_row', 'quoted_row', 'second_platform', 'held'])('row-ID collection rejects %s evidence', kind => { @@ -84,7 +87,7 @@ test.each(['foreign_row', 'quoted_row', 'second_platform', 'held'])('row-ID coll if (kind === 'second_platform') question.question = question.question.replace('?', ' or the Slack bot?'); call.answers = { [question.question]: kind === 'held' ? question.options[3]!.label : selected }; state.fingerprints = fromCalls(state.transcript.calls).fingerprints; - expect(accepts(state)).toBe(false); + expect(rowIdAccepts(state)).toBe(false); }); test('four candidate calls with five independent tabs meet the original floor', () => { diff --git a/test/e2e-tier-alignment.test.ts b/test/e2e-tier-alignment.test.ts index 13f0500cb..e5f830792 100644 --- a/test/e2e-tier-alignment.test.ts +++ b/test/e2e-tier-alignment.test.ts @@ -33,13 +33,13 @@ const TEST_DIR = import.meta.dir; // Both quote styles — a mechanical refactor to double quotes must not // silently drop a file from the invariant (fail-open is the defect class // this test exists to kill). -const SELF_GATE_RE = /EVALS_TIER\s*===\s*['"](gate|periodic)['"]/g; +const SELF_GATE_RE = /EVALS_TIER\s*===\s*['"](gate|periodic|marathon)['"]/g; // Consolidated gate helper (test/helpers/e2e-gate.ts). Both regexes stay // active: migrated files self-gate via `describeE2ETier('')` (or the // boolean form `e2eTierEnabled('')`), while stragglers still using the // raw predicate are caught by SELF_GATE_RE above. The tier argument maps to // the declared tier exactly like the raw predicate's tier literal did. -const HELPER_GATE_RE = /\b(?:describeE2ETier|e2eTierEnabled)\(\s*['"](gate|periodic)['"]/g; +const HELPER_GATE_RE = /\b(?:describeE2ETier|e2eTierEnabled)\(\s*['"](gate|periodic|marathon)['"]/g; /** * Ratchet, not amnesty (the contract KNOWN_MATRIX_GAPS pioneered before the diff --git a/test/helpers/office-hours-completion.ts b/test/helpers/office-hours-completion.ts index 5033d9e4a..36dfc21e4 100644 --- a/test/helpers/office-hours-completion.ts +++ b/test/helpers/office-hours-completion.ts @@ -91,6 +91,49 @@ function assignmentBody(markdown: string): string { || ''; } +/** + * Design-draft phase of the fixed fixture: the repo design carries every + * required section and an Assignment, and an independent Agent/Task opinion on + * RosterCheck preceded the Write that created it. The full workflow validator + * applies these same checks; the focused design-draft capture applies them alone. + */ +export function validateOfficeHoursDesignDraft( + evidence: Pick, + label = 'Office-hours design draft', +): { designPath: string; repoPath: string; firstDesignWrite: number } { + const fail = (message: string): never => { throw new Error(`${label}: ${message}`); }; + if (evidence.designContent === null) fail(`repo design is missing: ${evidence.designPath}`); + const design = evidence.designContent!; + for (const [section, names] of [ + ['Problem Statement', ['problem statement']], + ['Recommended Approach', ['recommended approach']], + ['Success Criteria', ['success criteria']], + ['What I noticed about how you think', ['what i noticed about how you think']], + ] as const) { + if (!substantive(sectionBody(design, [...names]))) fail(`repo design lacks substantive ${section}`); + } + if (!substantive(assignmentBody(design))) fail('repo design lacks a concrete Assignment'); + + // A cold-read opinion before the design exists is not the required spec + // review. The fixture promises an available Agent, so require an attempt + // that names this design even when the review subsequently fails. + const designPath = evidence.designPath.replace(/\\/g, '/'); + const repoPath = designPath.match(/(?:^|\/)(docs\/designs\/[^/]+\.md)$/)?.[1] ?? designPath; + const firstDesignWrite = evidence.toolCalls.findIndex(call => { + const writtenPath = String(call.input?.file_path ?? '').replace(/\\/g, '/').replace(/^\.\//, ''); + return call.tool === 'Write' && (writtenPath === designPath || writtenPath === repoPath); + }); + if (firstDesignWrite === -1) fail('no observed Write created the repo design'); + const opinion = evidence.toolCalls.slice(0, firstDesignWrite).some(call => { + if (!['Agent', 'Task'].includes(call.tool)) return false; + const prompt = `${String(call.input?.description ?? '')}\n${String(call.input?.prompt ?? '')}`; + return /\bRosterCheck\b/i.test(prompt) + && /\b(?:review|challenge|opinion|critique|perspective|steelman|advisor)\b|\bcold.read\b/i.test(prompt); + }); + if (!opinion) fail('no independent Agent/Task opinion on RosterCheck preceded the repo design Write'); + return { designPath, repoPath, firstDesignWrite }; +} + export function validateOfficeHoursCompletion(evidence: OfficeHoursCompletionEvidence): OfficeHoursReviewEvidence | null { const fail = (message: string): never => { throw new Error(`Office-hours completion: ${message}`); }; if (evidence.exitReason !== 'success') fail(`execution failed: ${evidence.exitReason}`); @@ -111,34 +154,8 @@ export function validateOfficeHoursCompletion(evidence: OfficeHoursCompletionEvi if (statuses.length !== 1 || statuses[0].trim().toUpperCase() !== 'APPROVED') { fail('repo design is not marked Status: APPROVED'); } - for (const [label, names] of [ - ['Problem Statement', ['problem statement']], - ['Recommended Approach', ['recommended approach']], - ['Success Criteria', ['success criteria']], - ['What I noticed about how you think', ['what i noticed about how you think']], - ] as const) { - if (!substantive(sectionBody(design, [...names]))) fail(`repo design lacks substantive ${label}`); - } - if (!substantive(assignmentBody(design))) fail('repo design lacks a concrete Assignment'); + const { designPath, repoPath, firstDesignWrite } = validateOfficeHoursDesignDraft(evidence, 'Office-hours completion'); if (!substantive(assignmentBody(evidence.output))) fail('REPORT.md lacks the Assignment'); - - // A cold-read opinion before the design exists is not the required spec - // review. The fixture promises an available Agent, so require an attempt - // that names this design even when the review subsequently fails. - const designPath = evidence.designPath.replace(/\\/g, '/'); - const repoPath = designPath.match(/(?:^|\/)(docs\/designs\/[^/]+\.md)$/)?.[1] ?? designPath; - const firstDesignWrite = evidence.toolCalls.findIndex(call => { - const writtenPath = String(call.input?.file_path ?? '').replace(/\\/g, '/').replace(/^\.\//, ''); - return call.tool === 'Write' && (writtenPath === designPath || writtenPath === repoPath); - }); - if (firstDesignWrite === -1) fail('no observed Write created the repo design'); - const opinion = evidence.toolCalls.slice(0, firstDesignWrite).some(call => { - if (!['Agent', 'Task'].includes(call.tool)) return false; - const prompt = `${String(call.input?.description ?? '')}\n${String(call.input?.prompt ?? '')}`; - return /\bRosterCheck\b/i.test(prompt) - && /\b(?:review|challenge|opinion|critique|perspective|steelman|advisor)\b|\bcold.read\b/i.test(prompt); - }); - if (!opinion) fail('no independent Agent/Task opinion on RosterCheck preceded the repo design Write'); const reviews = evidence.toolCalls.slice(firstDesignWrite + 1).filter(call => { if (!['Agent', 'Task'].includes(call.tool)) return false; const prompt = String(call.input?.prompt ?? '').replace(/\\/g, '/'); diff --git a/test/helpers/touchfiles-data.ts b/test/helpers/touchfiles-data.ts index 0a0780804..c4028d83c 100644 --- a/test/helpers/touchfiles-data.ts +++ b/test/helpers/touchfiles-data.ts @@ -1093,6 +1093,7 @@ export const E2E_TOUCHFILES: Record = { 'ship/SKILL.md', 'test/helpers/e2e-gate.ts'], 'office-hours-section-loading': [ 'office-hours/**', 'bin/gstack-office-hours-review', 'lib/office-hours-review.ts', 'lib/fs-atomic.ts', 'scripts/resolvers/review.ts', 'scripts/resolvers/sections.ts', 'scripts/gen-skill-docs.ts', 'test/helpers/carve-guards.ts', 'test/helpers/auq-sdk-capture.ts', 'test/helpers/office-hours-completion.ts', 'test/helpers/llm-judge.ts', 'test/helpers/session-runner.ts', 'test/skill-e2e-office-hours-section-loading.test.ts', 'test/helpers/agent-sdk-runner.ts', 'test/helpers/auq-native-capture.ts', 'test/helpers/auto-decision-state.ts', 'test/helpers/autoplan-artifact-digest.ts', 'test/helpers/autoplan-artifact-permission.ts', 'test/helpers/autoplan-artifact-recorder.ts', 'test/helpers/capture-parity-baseline.ts', 'test/helpers/claude-pty-runner.ts', 'test/helpers/dx-selected-navigation.ts', 'test/helpers/e2e-gate.ts', 'test/helpers/eng-cache-writer-decision.ts', 'test/helpers/hermetic-skill-runtime.ts', 'test/helpers/native-auto-decide.ts', 'test/helpers/owned-claude-transcript.ts', 'test/helpers/parity-harness.ts', 'test/helpers/plan-count-artifacts.ts', 'test/helpers/plan-count-file-permission.ts', 'test/helpers/plan-count-fixture.ts', 'test/helpers/plan-count-pending-exit.ts', 'test/helpers/plan-count-pending-question.ts', 'test/helpers/plan-count-transcript.ts', 'test/helpers/plan-floor-review.ts', 'test/helpers/plan-floor-target.ts', 'test/helpers/plan-scope-selection.ts', 'test/helpers/plan-seed-submission.ts', 'test/helpers/plan-skill-question-events.ts', 'test/helpers/plan-skill-question-hook-scope.ts', 'test/helpers/plan-skill-questions.ts', 'test/helpers/pty-screen.ts', 'test/helpers/pty-trust-dialog.ts', 'test/helpers/skill-census.ts'], + 'office-hours-design-draft': [ 'office-hours/**', 'lib/office-hours-review.ts', 'scripts/resolvers/review.ts', 'scripts/resolvers/sections.ts', 'scripts/gen-skill-docs.ts', 'test/helpers/carve-guards.ts', 'test/helpers/auq-sdk-capture.ts', 'test/helpers/office-hours-completion.ts', 'test/helpers/llm-judge.ts', 'test/helpers/session-runner.ts', 'test/skill-e2e-office-hours-design-draft.test.ts', 'test/helpers/agent-sdk-runner.ts', 'test/helpers/auq-native-capture.ts', 'test/helpers/auto-decision-state.ts', 'test/helpers/autoplan-artifact-digest.ts', 'test/helpers/autoplan-artifact-permission.ts', 'test/helpers/autoplan-artifact-recorder.ts', 'test/helpers/capture-parity-baseline.ts', 'test/helpers/claude-pty-runner.ts', 'test/helpers/dx-selected-navigation.ts', 'test/helpers/e2e-gate.ts', 'test/helpers/eng-cache-writer-decision.ts', 'test/helpers/hermetic-skill-runtime.ts', 'test/helpers/native-auto-decide.ts', 'test/helpers/owned-claude-transcript.ts', 'test/helpers/parity-harness.ts', 'test/helpers/plan-count-artifacts.ts', 'test/helpers/plan-count-file-permission.ts', 'test/helpers/plan-count-fixture.ts', 'test/helpers/plan-count-pending-exit.ts', 'test/helpers/plan-count-pending-question.ts', 'test/helpers/plan-count-transcript.ts', 'test/helpers/plan-floor-review.ts', 'test/helpers/plan-floor-target.ts', 'test/helpers/plan-scope-selection.ts', 'test/helpers/plan-seed-submission.ts', 'test/helpers/plan-skill-question-events.ts', 'test/helpers/plan-skill-question-hook-scope.ts', 'test/helpers/plan-skill-questions.ts', 'test/helpers/pty-screen.ts', 'test/helpers/pty-trust-dialog.ts', 'test/helpers/skill-census.ts'], 'plan-devex-peer-comparison-classification': [ @@ -1474,7 +1475,8 @@ export const E2E_TIERS: Record = { 'arm-benchmark-native-overbuild': 'periodic', 'arm-benchmark-crud-endpoint': 'periodic', 'arm-benchmark-bugfix-decoys': 'periodic', - 'office-hours-section-loading': 'periodic', // Full startup design/review/approval workflow + 'office-hours-section-loading': 'marathon', // Full startup design/review/approval workflow (1–3 real review rounds, ~20 min) + 'office-hours-design-draft': 'periodic', // Same interview through the design-creating Write (~5 min) 'plan-decision-classification': 'periodic', 'plan-devex-peer-comparison-classification': 'periodic', 'health-reporting': 'periodic', diff --git a/test/office-hours-budget.test.ts b/test/office-hours-budget.test.ts index d827713cb..1cc3bc28a 100644 --- a/test/office-hours-budget.test.ts +++ b/test/office-hours-budget.test.ts @@ -3,25 +3,39 @@ import * as fs from 'node:fs'; import * as os from 'node:os'; import * as path from 'node:path'; import { DEFAULT_SHARD_TIMEOUT_MS } from '../scripts/test-paid-shards'; +import { CAPTURE_LONG_MS } from './helpers/eval-budgets'; const casePath = path.join(import.meta.dir, 'skill-e2e-office-hours-section-loading.test.ts'); -// Evaluate the real case registration with inert test/describe functions. -// Imports are removed, and the captured paid callback is never invoked. -function registeredOptions(): { timeout: number; retry: number } { - const source = new Bun.Transpiler({ loader: 'ts' }).transformSync(fs.readFileSync(casePath, 'utf8')) +// Evaluate the real case registrations with inert test/describe functions. +// Imports are removed, and the captured paid callbacks are never invoked. +function registrations(file = casePath): Array<{ tier: string; options: unknown }> { + const source = new Bun.Transpiler({ loader: 'ts' }).transformSync(fs.readFileSync(file, 'utf8')) .replace(/^import\b[^;]*;\s*$/gm, ''); - const registrations: unknown[] = []; - new Function('test', 'describeE2ETier', source)( - (_name: string, _callback: unknown, options: unknown) => registrations.push(options), - () => (_name: string, register: () => void) => register(), + const found: Array<{ tier: string; options: unknown }> = []; + new Function('test', 'describeE2ETier', 'CAPTURE_LONG_MS', source)( + (_name: string, _callback: unknown, options: unknown) => found.at(-1)!.options = options, + (tier: string) => (_name: string, register: () => void) => { found.push({ tier, options: undefined }); register(); }, + CAPTURE_LONG_MS, ); - expect(registrations).toHaveLength(1); - return registrations[0] as { timeout: number; retry: number }; + return found; } +test('the full office-hours workflow is one marathon-tier registration', () => { + expect(registrations()).toEqual([{ tier: 'marathon', options: { timeout: 1_260_000, retry: 0 } }]); +}); + +test('the design-draft checkpoint is one periodic case inside the ordinary long capture budget', () => { + const draft = path.join(import.meta.dir, 'skill-e2e-office-hours-design-draft.test.ts'); + expect(registrations(draft)).toEqual([{ tier: 'periodic', options: CAPTURE_LONG_MS }]); + const source = fs.readFileSync(draft, 'utf8'); + expect(source).toContain('timeout: LONG_SECTION_CAPTURE_MS'); + expect(source).toContain('stop after the Write that saves the complete design'); + expect(source).toContain('Do not run the spec review, approval, relationship closing or handoff'); +}); + test('office-hours has one bounded attempt even under the paid runner CLI retry default', () => { - const options = registeredOptions(); + const options = registrations()[0]!.options as { timeout: number; retry: number }; expect(options).toEqual({ timeout: 1_260_000, retry: 0 }); expect(options.timeout + 120_000).toBeLessThanOrEqual(DEFAULT_SHARD_TIMEOUT_MS); diff --git a/test/office-hours-completion.test.ts b/test/office-hours-completion.test.ts index 2bc8f042c..05c8fc09d 100644 --- a/test/office-hours-completion.test.ts +++ b/test/office-hours-completion.test.ts @@ -3,7 +3,7 @@ import { describe, expect, test } from 'bun:test'; import * as fs from 'node:fs'; import * as path from 'node:path'; import * as os from 'node:os'; -import { validateOfficeHoursCompletion, validateOfficeHoursReviewerHandoffs, validateOfficeHoursReviewArtifacts, validateOfficeHoursReviewPreservation, validateOfficeHoursSpecSummary, type OfficeHoursCompletionEvidence } from './helpers/office-hours-completion'; +import { validateOfficeHoursCompletion, validateOfficeHoursDesignDraft, validateOfficeHoursReviewerHandoffs, validateOfficeHoursReviewArtifacts, validateOfficeHoursReviewPreservation, validateOfficeHoursSpecSummary, type OfficeHoursCompletionEvidence } from './helpers/office-hours-completion'; import { E2E_TOUCHFILES } from './helpers/touchfiles-data'; import { selectTests } from './helpers/test-selection'; @@ -72,6 +72,21 @@ describe('office-hours fixture completion', () => { expect(instructions).toContain('A failed command remains a failure'); }); + test('the design-draft checkpoint applies the full validator\'s design and opinion checks alone', () => { + const draftDesign = design.replace('Status: APPROVED', 'Status: DRAFT').replace(/## Reviewer Concerns[\s\S]*$/, ''); + const draft = { designPath, designContent: draftDesign, toolCalls: completed().toolCalls.slice(0, 2) }; + expect(validateOfficeHoursDesignDraft(draft)).toEqual({ designPath, repoPath: 'docs/designs/roster-check.md', firstDesignWrite: 1 }); + expect(() => validateOfficeHoursDesignDraft({ ...draft, toolCalls: draft.toolCalls.slice(1) })) + .toThrow('Office-hours design draft: no independent Agent/Task opinion'); + expect(() => validateOfficeHoursDesignDraft({ ...draft, toolCalls: [...draft.toolCalls].reverse() })) + .toThrow('no independent Agent/Task opinion'); + expect(() => validateOfficeHoursDesignDraft({ ...draft, designContent: draftDesign.replace(/## Success Criteria\n[^\n]+\n/, '') })) + .toThrow('repo design lacks substantive Success Criteria'); + expect(() => validateOfficeHoursDesignDraft({ ...draft, designContent: null })).toThrow('repo design is missing'); + expect(() => validateOfficeHoursCompletion({ ...completed(), toolCalls: completed().toolCalls.slice(1) })) + .toThrow('Office-hours completion: no independent Agent/Task opinion'); + }); + test('accepts a completed approved design with unresolved reviewer concerns', () => { const review = validateOfficeHoursCompletion(completed()); expect(review?.report).toBe(report); @@ -695,6 +710,7 @@ describe('office-hours completion eval selection', () => { test('office-hours source selects its dedicated workflow instead of the generic carve file', () => { const { selected } = selectTests(['office-hours/sections/design-and-handoff.md.tmpl'], E2E_TOUCHFILES); expect(selected).toContain('office-hours-section-loading'); + expect(selected).toContain('office-hours-design-draft'); expect(selected).not.toContain('carve-section-loading'); }); }); diff --git a/test/plan-scope-selection.test.ts b/test/plan-scope-selection.test.ts index 48ff7d47c..43e59e5ef 100644 --- a/test/plan-scope-selection.test.ts +++ b/test/plan-scope-selection.test.ts @@ -320,7 +320,7 @@ test('new scope route rejects stale, foreign, premature or unsuccessful evidence p => { p.tools[1]!.toolUseId = 'unrelated'; }, p => { p.tools[1]!.sessionId = 'foreign'; }, p => { p.tools.splice(1, 1); }, - p => { p.tools.push(structuredClone(p.tools[1]!)); }, + p => { p.tools.push(structuredClone(p.tools[1]!) as never); }, p => { p.tools[2]!.timestamp = new Date(p.opts.commandStartedAt + 1).toISOString(); }, p => { p.tools[2]!.timestamp = new Date(Date.parse(p.tools[1]!.timestamp) - 1).toISOString(); }, p => { p.tools[2]!.timestamp = 'unknown'; }, diff --git a/test/qa-supervision-selection.test.ts b/test/qa-supervision-selection.test.ts index 959d9b77b..fcf38d955 100644 --- a/test/qa-supervision-selection.test.ts +++ b/test/qa-supervision-selection.test.ts @@ -5,7 +5,7 @@ const ptyIds = [ 'plan-ceo-review-plan-mode', 'plan-eng-review-plan-mode', 'plan-design-review-plan-mode', 'plan-devex-review-plan-mode', 'plan-mode-no-op', 'office-hours-auto-mode', 'auto-decide-preserved', 'plan-ceo-mode-routing', 'plan-design-with-ui-scope', 'plan-eng-finding-floor', - 'auq-format-gate', 'carve-section-loading', 'office-hours-section-loading', 'plan-ceo-section-loading', 'ship-section-loading', + 'auq-format-gate', 'carve-section-loading', 'office-hours-section-loading', 'office-hours-design-draft', 'plan-ceo-section-loading', 'ship-section-loading', 'plan-ceo-finding-floor', 'plan-design-finding-floor', 'plan-devex-finding-floor', 'plan-eng-multi-finding-batching', 'plan-ceo-split-overflow', ].sort(); diff --git a/test/skill-e2e-office-hours-design-draft.test.ts b/test/skill-e2e-office-hours-design-draft.test.ts new file mode 100644 index 000000000..de211fccf --- /dev/null +++ b/test/skill-e2e-office-hours-design-draft.test.ts @@ -0,0 +1,68 @@ +/** + * Office-hours design-draft checkpoint (periodic). The fixed startup interview + * from CARVE_GUARDS['office-hours'] runs only through the Write that creates the + * design (observed at 269s of the full workflow in run 36385945043). It applies + * the full workflow's design-draft checks, required section reads and + * launch/skill-read guards. The complete review, approval and handoff workflow + * is marathon tier in skill-e2e-office-hours-section-loading.test.ts. + */ +import { test, expect } from 'bun:test'; +import * as fs from 'node:fs'; +import * as path from 'node:path'; +import { describeE2ETier } from './helpers/e2e-gate'; +import { CAPTURE_LONG_MS } from './helpers/eval-budgets'; +import { setupSkillDir, skillFromWorktree, captureSectionReads, LONG_SECTION_CAPTURE_MS } from './helpers/auq-sdk-capture'; +import { CARVE_GUARDS } from './helpers/carve-guards'; +import { validateOfficeHoursDesignDraft } from './helpers/office-hours-completion'; + +const describePeriodic = describeE2ETier('periodic'); + +describePeriodic('/office-hours design-draft checkpoint (periodic)', () => { + test('the startup interview reads its sections and writes a complete design after an independent opinion', async () => { + const guard = CARVE_GUARDS['office-hours']; + const answersStart = guard.scenario.indexOf('This is a synthetic interview fixture'); + const answersEnd = guard.scenario.indexOf('Use the available Agent tool'); + if (answersStart < 0 || answersEnd < answersStart) { + throw new Error('fixture drift: CARVE_GUARDS office-hours scenario no longer has its interview block'); + } + const { skillMd, sectionsFrom } = skillFromWorktree(guard.skill); + const planDir = setupSkillDir({ + skillName: guard.skill, skillMd, sectionsFrom, tmpPrefix: 'gstack-office-hours-draft-', + }); + const capture = await captureSectionReads({ + planDir, skillName: guard.skill, + scenario: `Run office hours in STARTUP mode through the diagnostic, premise challenge, +alternatives, independent opinion and the design document. +${guard.scenario.slice(answersStart, answersEnd).trim()} + +Use the available Agent tool for the independent opinion when the workflow calls +for it. It is read-only. An automated test is not a reason to skip it. This fixture +checks a separate real tool call: obtain the independent opinion on RosterCheck +before writing the design. A self-authored "second opinion" is not a subagent result. +Save the design to docs/designs/roster-check.md in this fixture; keep all artifacts +inside the fixture. + +Checkpoint for this run: stop after the Write that saves the complete design +document. Do not run the spec review, approval, relationship closing or handoff; +the full workflow covers them separately. Then write REPORT.md with the design path +and a one-line outcome for each phase you completed.`, + testName: 'office-hours-design-draft', runId: `office-hours-design-draft-${process.env.EVALS_RUN_ID ?? 'local'}`, + timeout: LONG_SECTION_CAPTURE_MS, + }); + const designPath = path.join(planDir, 'docs/designs/roster-check.md'); + expect(capture.exitReason).toBe('success'); + validateOfficeHoursDesignDraft({ + designPath, toolCalls: capture.toolCalls, + designContent: fs.existsSync(designPath) ? fs.readFileSync(designPath, 'utf-8') : null, + }); + const missing = guard.requiredReads.filter(section => !capture.readSections.has(section)); + expect({ reportProduced: capture.reportProduced, read: [...capture.readSections], missing }).toEqual({ + reportProduced: true, read: expect.any(Array), missing: [], + }); + expect(capture.toolCalls.filter(call => call.tool === 'Skill')).toEqual([]); + const ownSkillPath = path.join(planDir, guard.skill, 'SKILL.md'); + expect(capture.toolCalls.filter(call => call.tool === 'Read' + && /(?:^|\/)SKILL\.md$/.test(String(call.input?.file_path ?? '')) + && path.resolve(planDir, call.input.file_path) !== ownSkillPath)).toEqual([]); + }, CAPTURE_LONG_MS); +}); diff --git a/test/skill-e2e-office-hours-section-loading.test.ts b/test/skill-e2e-office-hours-section-loading.test.ts index 43daf4dce..af1f4f238 100644 --- a/test/skill-e2e-office-hours-section-loading.test.ts +++ b/test/skill-e2e-office-hours-section-loading.test.ts @@ -1,7 +1,12 @@ /** - * Full office-hours startup workflow, isolated from the generic carve shard. + * Office-hours startup workflow, isolated from the generic carve shard. * A fixed interview exercises real opinion/design/review/approval/handoff work. * Free completion regressions live in office-hours-completion.test.ts. + * + * This full start-to-finish workflow (1–3 real spec-review rounds, ~20 min) is + * marathon tier. skill-e2e-office-hours-design-draft.test.ts runs the same + * interview only through the checkpoint that creates the design (~5 min) in + * the periodic lane. */ import { test, expect } from 'bun:test'; import * as fs from 'node:fs'; @@ -11,7 +16,7 @@ import { setupSkillDir, skillFromWorktree, captureSectionReads } from './helpers import { CARVE_GUARDS } from './helpers/carve-guards'; import { validateOfficeHoursCompletion, validateOfficeHoursReviewerHandoffs, validateOfficeHoursReviewArtifacts, validateOfficeHoursReviewPreservation } from './helpers/office-hours-completion'; -const describeE2E = describeE2ETier('periodic'); +const describeMarathon = describeE2ETier('marathon'); const runId = `office-hours-section-loading-${process.env.EVALS_RUN_ID ?? 'local'}`; // Full startup diagnosis + outside opinion + up to three spec reviews exceeds @@ -25,7 +30,7 @@ const runId = `office-hours-section-loading-${process.env.EVALS_RUN_ID ?? 'local const OFFICE_HOURS_CAPTURE_MS = 1_200_000; const OFFICE_HOURS_TEST_MS = 1_260_000; -describeE2E('/office-hours full section-loading workflow (periodic)', () => { +describeMarathon('/office-hours full section-loading workflow (marathon)', () => { test('a real startup review reads its sections and completes the approved design and handoff', async () => { const guard = CARVE_GUARDS['office-hours']; const { skillMd, sectionsFrom } = skillFromWorktree(guard.skill); diff --git a/test/touchfiles.test.ts b/test/touchfiles.test.ts index b476fd45a..3d72c708e 100644 --- a/test/touchfiles.test.ts +++ b/test/touchfiles.test.ts @@ -510,7 +510,7 @@ describe('TOUCHFILES completeness', () => { }); test('E2E_TIERS only contains valid tier values', () => { - const validTiers = ['gate', 'periodic']; + const validTiers = ['gate', 'periodic', 'marathon']; for (const [name, tier] of Object.entries(E2E_TIERS)) { if (!validTiers.includes(tier)) { throw new Error(`E2E_TIERS['${name}'] has invalid tier '${tier}'. Valid: ${validTiers.join(', ')}`);