test: move the full office-hours workflow to marathon; add a periodic design-draft checkpoint

The full startup workflow runs 1–3 real spec-review rounds (~280s each) and
hit its 1200s capture in run 36385945043 at finalize. Review depth is the
product's loop, so the case cannot fit a blocking lane without cutting
rounds. It is now marathon tier with every assertion unchanged.

skill-e2e-office-hours-design-draft.test.ts (periodic) runs the same fixed
interview only through the Write that creates the design (269s in that run)
and applies the full validator's design-draft checks, the required section
reads and the launch/foreign-skill-read guards. validateOfficeHoursDesignDraft
is extracted from validateOfficeHoursCompletion, which still applies it.

Selection: office-hours-design-draft is registered periodic; the marathon-only
file is already excluded from the gate and periodic plans by the B5 planner
rule. Tier-alignment regexes and the valid-tier check accept 'marathon'.
A type-only cast in plan-scope-selection.test.ts removes a diagnostic whose
union print order made the ratchet identity unstable; baseline tightened.
This commit is contained in:
garrytan committed 2026-09-29 15:29:46 +00:00
1 parent a41cdb7d7d
commit 4973f96d98
12 files changed
+180 -57

No files matched your search

+10 -7
View File
@@ -2,7 +2,8 @@ import { expect, test } from 'bun:test';
import * as fs from 'node:fs';
import * as os from 'node:os';
import * as path from 'node:path';
import { nativePlanCallFingerprint } from './helpers/claude-pty-runner';
import { nativePlanCallFingerprint, type AskUserQuestionFingerprint } from './helpers/claude-pty-runner';
import type { NativeQuestion } from './helpers/plan-skill-questions';
import type { NativePlanQuestionCall, PlanCountTranscript } from './helpers/plan-count-transcript';
import { ceoSplitCandidate, ceoSplitDecisionFingerprints, isCeoSplitCandidateCall, isCeoSplitCollectionComplete } from './helpers/ceo-split-question-policy';
import captured from './fixtures/ceo-split-collection-0bcd.json';
@@ -54,26 +55,28 @@ test.each([0, 1, 2, 3, 4, 5, 6])('the exact original %i-call prefix waits for th
// Run 36385945043: the skill cited ledger row IDs ("D2.1 — R-E1: …") and offered
// a fourth "Hold, discuss first" option. No candidate was recognized, so collection
// never stopped and the attempt ran the whole review (1302s) after the E5 ACK.
function rowIdCapture() {
const calls = structuredClone(rowIds.calls) as NativePlanQuestionCall[];
function rowIdCapture(): { transcript: PlanCountTranscript; fingerprints: AskUserQuestionFingerprint[] } {
const calls = structuredClone(rowIds.calls) as unknown as NativePlanQuestionCall[];
const transcript: PlanCountTranscript = { status: 'ready', calls, assistantMessages: [] };
const fingerprints = rowIds.fingerprints.map((fp, index) => ({ ...structuredClone(fp), nativeCall: calls[index]! }));
return { transcript, fingerprints };
}
const rowIdAccepts = (state: ReturnType<typeof rowIdCapture>) => isCeoSplitCollectionComplete(state.transcript, state.fingerprints);
test('ledger row-ID candidate questions from run 36385945043 finish collection at the E5 ACK', () => {
const state = rowIdCapture();
expect(rowIds.provenance.originalOutcome).toBe('completion_summary');
expect(rowIds.provenance.originalReviewCount).toBe(0);
expect(state.transcript.calls.at(-1)!.answeredAt).toBe(rowIds.provenance.completeAt);
expect(state.transcript.calls.map(call => ceoSplitCandidate(call.questions[0]!))).toEqual([null, 'E1', 'E2', 'E3', 'E4', 'E5']);
expect(state.transcript.calls.map(call => ceoSplitCandidate(call.questions[0] as NativeQuestion)))
.toEqual([null, 'E1', 'E2', 'E3', 'E4', 'E5']);
expect(state.fingerprints.map(isCeoSplitCandidateCall)).toEqual([false, true, true, true, true, true]);
for (let length = 0; length < 6; length++) {
const prefix = rowIdCapture();
prefix.transcript.calls.length = length; prefix.fingerprints.length = length;
expect(accepts(prefix)).toBe(false);
expect(rowIdAccepts(prefix)).toBe(false);
}
expect(accepts(state)).toBe(true);
expect(rowIdAccepts(state)).toBe(true);
});
test.each(['foreign_row', 'quoted_row', 'second_platform', 'held'])('row-ID collection rejects %s evidence', kind => {
@@ -84,7 +87,7 @@ test.each(['foreign_row', 'quoted_row', 'second_platform', 'held'])('row-ID coll
if (kind === 'second_platform') question.question = question.question.replace('?', ' or the Slack bot?');
call.answers = { [question.question]: kind === 'held' ? question.options[3]!.label : selected };
state.fingerprints = fromCalls(state.transcript.calls).fingerprints;
expect(accepts(state)).toBe(false);
expect(rowIdAccepts(state)).toBe(false);
});
test('four candidate calls with five independent tabs meet the original floor', () => {