test: move the full office-hours workflow to marathon; add a periodic design-draft checkpoint

The full startup workflow runs 1–3 real spec-review rounds (~280s each) and
hit its 1200s capture in run 36385945043 at finalize. Review depth is the
product's loop, so the case cannot fit a blocking lane without cutting
rounds. It is now marathon tier with every assertion unchanged.

skill-e2e-office-hours-design-draft.test.ts (periodic) runs the same fixed
interview only through the Write that creates the design (269s in that run)
and applies the full validator's design-draft checks, the required section
reads and the launch/foreign-skill-read guards. validateOfficeHoursDesignDraft
is extracted from validateOfficeHoursCompletion, which still applies it.

Selection: office-hours-design-draft is registered periodic; the marathon-only
file is already excluded from the gate and periodic plans by the B5 planner
rule. Tier-alignment regexes and the valid-tier check accept 'marathon'.
A type-only cast in plan-scope-selection.test.ts removes a diagnostic whose
union print order made the ratchet identity unstable; baseline tightened.
This commit is contained in:
garrytan committed 2026-09-29 15:29:46 +00:00
1 parent a41cdb7d7d
commit 4973f96d98
12 files changed
+180 -57

No files matched your search

+10 -7
View File
@@ -2,7 +2,8 @@ import { expect, test } from 'bun:test';
import * as fs from 'node:fs';
import * as os from 'node:os';
import * as path from 'node:path';
import { nativePlanCallFingerprint } from './helpers/claude-pty-runner';
import { nativePlanCallFingerprint, type AskUserQuestionFingerprint } from './helpers/claude-pty-runner';
import type { NativeQuestion } from './helpers/plan-skill-questions';
import type { NativePlanQuestionCall, PlanCountTranscript } from './helpers/plan-count-transcript';
import { ceoSplitCandidate, ceoSplitDecisionFingerprints, isCeoSplitCandidateCall, isCeoSplitCollectionComplete } from './helpers/ceo-split-question-policy';
import captured from './fixtures/ceo-split-collection-0bcd.json';
@@ -54,26 +55,28 @@ test.each([0, 1, 2, 3, 4, 5, 6])('the exact original %i-call prefix waits for th
// Run 36385945043: the skill cited ledger row IDs ("D2.1 — R-E1: …") and offered
// a fourth "Hold, discuss first" option. No candidate was recognized, so collection
// never stopped and the attempt ran the whole review (1302s) after the E5 ACK.
function rowIdCapture() {
const calls = structuredClone(rowIds.calls) as NativePlanQuestionCall[];
function rowIdCapture(): { transcript: PlanCountTranscript; fingerprints: AskUserQuestionFingerprint[] } {
const calls = structuredClone(rowIds.calls) as unknown as NativePlanQuestionCall[];
const transcript: PlanCountTranscript = { status: 'ready', calls, assistantMessages: [] };
const fingerprints = rowIds.fingerprints.map((fp, index) => ({ ...structuredClone(fp), nativeCall: calls[index]! }));
return { transcript, fingerprints };
}
const rowIdAccepts = (state: ReturnType<typeof rowIdCapture>) => isCeoSplitCollectionComplete(state.transcript, state.fingerprints);
test('ledger row-ID candidate questions from run 36385945043 finish collection at the E5 ACK', () => {
const state = rowIdCapture();
expect(rowIds.provenance.originalOutcome).toBe('completion_summary');
expect(rowIds.provenance.originalReviewCount).toBe(0);
expect(state.transcript.calls.at(-1)!.answeredAt).toBe(rowIds.provenance.completeAt);
expect(state.transcript.calls.map(call => ceoSplitCandidate(call.questions[0]!))).toEqual([null, 'E1', 'E2', 'E3', 'E4', 'E5']);
expect(state.transcript.calls.map(call => ceoSplitCandidate(call.questions[0] as NativeQuestion)))
.toEqual([null, 'E1', 'E2', 'E3', 'E4', 'E5']);
expect(state.fingerprints.map(isCeoSplitCandidateCall)).toEqual([false, true, true, true, true, true]);
for (let length = 0; length < 6; length++) {
const prefix = rowIdCapture();
prefix.transcript.calls.length = length; prefix.fingerprints.length = length;
expect(accepts(prefix)).toBe(false);
expect(rowIdAccepts(prefix)).toBe(false);
}
expect(accepts(state)).toBe(true);
expect(rowIdAccepts(state)).toBe(true);
});
test.each(['foreign_row', 'quoted_row', 'second_platform', 'held'])('row-ID collection rejects %s evidence', kind => {
@@ -84,7 +87,7 @@ test.each(['foreign_row', 'quoted_row', 'second_platform', 'held'])('row-ID coll
if (kind === 'second_platform') question.question = question.question.replace('?', ' or the Slack bot?');
call.answers = { [question.question]: kind === 'held' ? question.options[3]!.label : selected };
state.fingerprints = fromCalls(state.transcript.calls).fingerprints;
expect(accepts(state)).toBe(false);
expect(rowIdAccepts(state)).toBe(false);
});
test('four candidate calls with five independent tabs meet the original floor', () => {
+2 -2
View File
@@ -33,13 +33,13 @@ const TEST_DIR = import.meta.dir;
// Both quote styles — a mechanical refactor to double quotes must not
// silently drop a file from the invariant (fail-open is the defect class
// this test exists to kill).
const SELF_GATE_RE = /EVALS_TIER\s*===\s*['"](gate|periodic)['"]/g;
const SELF_GATE_RE = /EVALS_TIER\s*===\s*['"](gate|periodic|marathon)['"]/g;
// Consolidated gate helper (test/helpers/e2e-gate.ts). Both regexes stay
// active: migrated files self-gate via `describeE2ETier('<tier>')` (or the
// boolean form `e2eTierEnabled('<tier>')`), while stragglers still using the
// raw predicate are caught by SELF_GATE_RE above. The tier argument maps to
// the declared tier exactly like the raw predicate's tier literal did.
const HELPER_GATE_RE = /\b(?:describeE2ETier|e2eTierEnabled)\(\s*['"](gate|periodic)['"]/g;
const HELPER_GATE_RE = /\b(?:describeE2ETier|e2eTierEnabled)\(\s*['"](gate|periodic|marathon)['"]/g;
/**
* Ratchet, not amnesty (the contract KNOWN_MATRIX_GAPS pioneered before the
+44 -27
View File
@@ -91,6 +91,49 @@ function assignmentBody(markdown: string): string {
|| '';
}
/**
* Design-draft phase of the fixed fixture: the repo design carries every
* required section and an Assignment, and an independent Agent/Task opinion on
* RosterCheck preceded the Write that created it. The full workflow validator
* applies these same checks; the focused design-draft capture applies them alone.
*/
export function validateOfficeHoursDesignDraft(
evidence: Pick<OfficeHoursCompletionEvidence, 'designPath' | 'designContent' | 'toolCalls'>,
label = 'Office-hours design draft',
): { designPath: string; repoPath: string; firstDesignWrite: number } {
const fail = (message: string): never => { throw new Error(`${label}: ${message}`); };
if (evidence.designContent === null) fail(`repo design is missing: ${evidence.designPath}`);
const design = evidence.designContent!;
for (const [section, names] of [
['Problem Statement', ['problem statement']],
['Recommended Approach', ['recommended approach']],
['Success Criteria', ['success criteria']],
['What I noticed about how you think', ['what i noticed about how you think']],
] as const) {
if (!substantive(sectionBody(design, [...names]))) fail(`repo design lacks substantive ${section}`);
}
if (!substantive(assignmentBody(design))) fail('repo design lacks a concrete Assignment');
// A cold-read opinion before the design exists is not the required spec
// review. The fixture promises an available Agent, so require an attempt
// that names this design even when the review subsequently fails.
const designPath = evidence.designPath.replace(/\\/g, '/');
const repoPath = designPath.match(/(?:^|\/)(docs\/designs\/[^/]+\.md)$/)?.[1] ?? designPath;
const firstDesignWrite = evidence.toolCalls.findIndex(call => {
const writtenPath = String(call.input?.file_path ?? '').replace(/\\/g, '/').replace(/^\.\//, '');
return call.tool === 'Write' && (writtenPath === designPath || writtenPath === repoPath);
});
if (firstDesignWrite === -1) fail('no observed Write created the repo design');
const opinion = evidence.toolCalls.slice(0, firstDesignWrite).some(call => {
if (!['Agent', 'Task'].includes(call.tool)) return false;
const prompt = `${String(call.input?.description ?? '')}\n${String(call.input?.prompt ?? '')}`;
return /\bRosterCheck\b/i.test(prompt)
&& /\b(?:review|challenge|opinion|critique|perspective|steelman|advisor)\b|\bcold.read\b/i.test(prompt);
});
if (!opinion) fail('no independent Agent/Task opinion on RosterCheck preceded the repo design Write');
return { designPath, repoPath, firstDesignWrite };
}
export function validateOfficeHoursCompletion(evidence: OfficeHoursCompletionEvidence): OfficeHoursReviewEvidence | null {
const fail = (message: string): never => { throw new Error(`Office-hours completion: ${message}`); };
if (evidence.exitReason !== 'success') fail(`execution failed: ${evidence.exitReason}`);
@@ -111,34 +154,8 @@ export function validateOfficeHoursCompletion(evidence: OfficeHoursCompletionEvi
if (statuses.length !== 1 || statuses[0].trim().toUpperCase() !== 'APPROVED') {
fail('repo design is not marked Status: APPROVED');
}
for (const [label, names] of [
['Problem Statement', ['problem statement']],
['Recommended Approach', ['recommended approach']],
['Success Criteria', ['success criteria']],
['What I noticed about how you think', ['what i noticed about how you think']],
] as const) {
if (!substantive(sectionBody(design, [...names]))) fail(`repo design lacks substantive ${label}`);
}
if (!substantive(assignmentBody(design))) fail('repo design lacks a concrete Assignment');
const { designPath, repoPath, firstDesignWrite } = validateOfficeHoursDesignDraft(evidence, 'Office-hours completion');
if (!substantive(assignmentBody(evidence.output))) fail('REPORT.md lacks the Assignment');
// A cold-read opinion before the design exists is not the required spec
// review. The fixture promises an available Agent, so require an attempt
// that names this design even when the review subsequently fails.
const designPath = evidence.designPath.replace(/\\/g, '/');
const repoPath = designPath.match(/(?:^|\/)(docs\/designs\/[^/]+\.md)$/)?.[1] ?? designPath;
const firstDesignWrite = evidence.toolCalls.findIndex(call => {
const writtenPath = String(call.input?.file_path ?? '').replace(/\\/g, '/').replace(/^\.\//, '');
return call.tool === 'Write' && (writtenPath === designPath || writtenPath === repoPath);
});
if (firstDesignWrite === -1) fail('no observed Write created the repo design');
const opinion = evidence.toolCalls.slice(0, firstDesignWrite).some(call => {
if (!['Agent', 'Task'].includes(call.tool)) return false;
const prompt = `${String(call.input?.description ?? '')}\n${String(call.input?.prompt ?? '')}`;
return /\bRosterCheck\b/i.test(prompt)
&& /\b(?:review|challenge|opinion|critique|perspective|steelman|advisor)\b|\bcold.read\b/i.test(prompt);
});
if (!opinion) fail('no independent Agent/Task opinion on RosterCheck preceded the repo design Write');
const reviews = evidence.toolCalls.slice(firstDesignWrite + 1).filter(call => {
if (!['Agent', 'Task'].includes(call.tool)) return false;
const prompt = String(call.input?.prompt ?? '').replace(/\\/g, '/');
+3 -1
View File
@@ -1093,6 +1093,7 @@ export const E2E_TOUCHFILES: Record<string, string[]> = {
'ship/SKILL.md', 'test/helpers/e2e-gate.ts'],
'office-hours-section-loading': [ 'office-hours/**', 'bin/gstack-office-hours-review', 'lib/office-hours-review.ts', 'lib/fs-atomic.ts', 'scripts/resolvers/review.ts', 'scripts/resolvers/sections.ts', 'scripts/gen-skill-docs.ts', 'test/helpers/carve-guards.ts', 'test/helpers/auq-sdk-capture.ts', 'test/helpers/office-hours-completion.ts', 'test/helpers/llm-judge.ts', 'test/helpers/session-runner.ts', 'test/skill-e2e-office-hours-section-loading.test.ts', 'test/helpers/agent-sdk-runner.ts', 'test/helpers/auq-native-capture.ts', 'test/helpers/auto-decision-state.ts', 'test/helpers/autoplan-artifact-digest.ts', 'test/helpers/autoplan-artifact-permission.ts', 'test/helpers/autoplan-artifact-recorder.ts', 'test/helpers/capture-parity-baseline.ts', 'test/helpers/claude-pty-runner.ts', 'test/helpers/dx-selected-navigation.ts', 'test/helpers/e2e-gate.ts', 'test/helpers/eng-cache-writer-decision.ts', 'test/helpers/hermetic-skill-runtime.ts', 'test/helpers/native-auto-decide.ts', 'test/helpers/owned-claude-transcript.ts', 'test/helpers/parity-harness.ts', 'test/helpers/plan-count-artifacts.ts', 'test/helpers/plan-count-file-permission.ts', 'test/helpers/plan-count-fixture.ts', 'test/helpers/plan-count-pending-exit.ts', 'test/helpers/plan-count-pending-question.ts', 'test/helpers/plan-count-transcript.ts', 'test/helpers/plan-floor-review.ts', 'test/helpers/plan-floor-target.ts', 'test/helpers/plan-scope-selection.ts', 'test/helpers/plan-seed-submission.ts', 'test/helpers/plan-skill-question-events.ts', 'test/helpers/plan-skill-question-hook-scope.ts', 'test/helpers/plan-skill-questions.ts', 'test/helpers/pty-screen.ts', 'test/helpers/pty-trust-dialog.ts', 'test/helpers/skill-census.ts'],
'office-hours-design-draft': [ 'office-hours/**', 'lib/office-hours-review.ts', 'scripts/resolvers/review.ts', 'scripts/resolvers/sections.ts', 'scripts/gen-skill-docs.ts', 'test/helpers/carve-guards.ts', 'test/helpers/auq-sdk-capture.ts', 'test/helpers/office-hours-completion.ts', 'test/helpers/llm-judge.ts', 'test/helpers/session-runner.ts', 'test/skill-e2e-office-hours-design-draft.test.ts', 'test/helpers/agent-sdk-runner.ts', 'test/helpers/auq-native-capture.ts', 'test/helpers/auto-decision-state.ts', 'test/helpers/autoplan-artifact-digest.ts', 'test/helpers/autoplan-artifact-permission.ts', 'test/helpers/autoplan-artifact-recorder.ts', 'test/helpers/capture-parity-baseline.ts', 'test/helpers/claude-pty-runner.ts', 'test/helpers/dx-selected-navigation.ts', 'test/helpers/e2e-gate.ts', 'test/helpers/eng-cache-writer-decision.ts', 'test/helpers/hermetic-skill-runtime.ts', 'test/helpers/native-auto-decide.ts', 'test/helpers/owned-claude-transcript.ts', 'test/helpers/parity-harness.ts', 'test/helpers/plan-count-artifacts.ts', 'test/helpers/plan-count-file-permission.ts', 'test/helpers/plan-count-fixture.ts', 'test/helpers/plan-count-pending-exit.ts', 'test/helpers/plan-count-pending-question.ts', 'test/helpers/plan-count-transcript.ts', 'test/helpers/plan-floor-review.ts', 'test/helpers/plan-floor-target.ts', 'test/helpers/plan-scope-selection.ts', 'test/helpers/plan-seed-submission.ts', 'test/helpers/plan-skill-question-events.ts', 'test/helpers/plan-skill-question-hook-scope.ts', 'test/helpers/plan-skill-questions.ts', 'test/helpers/pty-screen.ts', 'test/helpers/pty-trust-dialog.ts', 'test/helpers/skill-census.ts'],
'plan-devex-peer-comparison-classification': [
@@ -1474,7 +1475,8 @@ export const E2E_TIERS: Record<string, 'gate' | 'periodic' | 'marathon'> = {
'arm-benchmark-native-overbuild': 'periodic',
'arm-benchmark-crud-endpoint': 'periodic',
'arm-benchmark-bugfix-decoys': 'periodic',
'office-hours-section-loading': 'periodic', // Full startup design/review/approval workflow
'office-hours-section-loading': 'marathon', // Full startup design/review/approval workflow (1–3 real review rounds, ~20 min)
'office-hours-design-draft': 'periodic', // Same interview through the design-creating Write (~5 min)
'plan-decision-classification': 'periodic',
'plan-devex-peer-comparison-classification': 'periodic',
'health-reporting': 'periodic',
+25 -11
View File
@@ -3,25 +3,39 @@ import * as fs from 'node:fs';
import * as os from 'node:os';
import * as path from 'node:path';
import { DEFAULT_SHARD_TIMEOUT_MS } from '../scripts/test-paid-shards';
import { CAPTURE_LONG_MS } from './helpers/eval-budgets';
const casePath = path.join(import.meta.dir, 'skill-e2e-office-hours-section-loading.test.ts');
// Evaluate the real case registration with inert test/describe functions.
// Imports are removed, and the captured paid callback is never invoked.
function registeredOptions(): { timeout: number; retry: number } {
const source = new Bun.Transpiler({ loader: 'ts' }).transformSync(fs.readFileSync(casePath, 'utf8'))
// Evaluate the real case registrations with inert test/describe functions.
// Imports are removed, and the captured paid callbacks are never invoked.
function registrations(file = casePath): Array<{ tier: string; options: unknown }> {
const source = new Bun.Transpiler({ loader: 'ts' }).transformSync(fs.readFileSync(file, 'utf8'))
.replace(/^import\b[^;]*;\s*$/gm, '');
const registrations: unknown[] = [];
new Function('test', 'describeE2ETier', source)(
(_name: string, _callback: unknown, options: unknown) => registrations.push(options),
() => (_name: string, register: () => void) => register(),
const found: Array<{ tier: string; options: unknown }> = [];
new Function('test', 'describeE2ETier', 'CAPTURE_LONG_MS', source)(
(_name: string, _callback: unknown, options: unknown) => found.at(-1)!.options = options,
(tier: string) => (_name: string, register: () => void) => { found.push({ tier, options: undefined }); register(); },
CAPTURE_LONG_MS,
);
expect(registrations).toHaveLength(1);
return registrations[0] as { timeout: number; retry: number };
return found;
}
test('the full office-hours workflow is one marathon-tier registration', () => {
expect(registrations()).toEqual([{ tier: 'marathon', options: { timeout: 1_260_000, retry: 0 } }]);
});
test('the design-draft checkpoint is one periodic case inside the ordinary long capture budget', () => {
const draft = path.join(import.meta.dir, 'skill-e2e-office-hours-design-draft.test.ts');
expect(registrations(draft)).toEqual([{ tier: 'periodic', options: CAPTURE_LONG_MS }]);
const source = fs.readFileSync(draft, 'utf8');
expect(source).toContain('timeout: LONG_SECTION_CAPTURE_MS');
expect(source).toContain('stop after the Write that saves the complete design');
expect(source).toContain('Do not run the spec review, approval, relationship closing or handoff');
});
test('office-hours has one bounded attempt even under the paid runner CLI retry default', () => {
const options = registeredOptions();
const options = registrations()[0]!.options as { timeout: number; retry: number };
expect(options).toEqual({ timeout: 1_260_000, retry: 0 });
expect(options.timeout + 120_000).toBeLessThanOrEqual(DEFAULT_SHARD_TIMEOUT_MS);
+17 -1
View File
@@ -3,7 +3,7 @@ import { describe, expect, test } from 'bun:test';
import * as fs from 'node:fs';
import * as path from 'node:path';
import * as os from 'node:os';
import { validateOfficeHoursCompletion, validateOfficeHoursReviewerHandoffs, validateOfficeHoursReviewArtifacts, validateOfficeHoursReviewPreservation, validateOfficeHoursSpecSummary, type OfficeHoursCompletionEvidence } from './helpers/office-hours-completion';
import { validateOfficeHoursCompletion, validateOfficeHoursDesignDraft, validateOfficeHoursReviewerHandoffs, validateOfficeHoursReviewArtifacts, validateOfficeHoursReviewPreservation, validateOfficeHoursSpecSummary, type OfficeHoursCompletionEvidence } from './helpers/office-hours-completion';
import { E2E_TOUCHFILES } from './helpers/touchfiles-data';
import { selectTests } from './helpers/test-selection';
@@ -72,6 +72,21 @@ describe('office-hours fixture completion', () => {
expect(instructions).toContain('A failed command remains a failure');
});
test('the design-draft checkpoint applies the full validator\'s design and opinion checks alone', () => {
const draftDesign = design.replace('Status: APPROVED', 'Status: DRAFT').replace(/## Reviewer Concerns[\s\S]*$/, '');
const draft = { designPath, designContent: draftDesign, toolCalls: completed().toolCalls.slice(0, 2) };
expect(validateOfficeHoursDesignDraft(draft)).toEqual({ designPath, repoPath: 'docs/designs/roster-check.md', firstDesignWrite: 1 });
expect(() => validateOfficeHoursDesignDraft({ ...draft, toolCalls: draft.toolCalls.slice(1) }))
.toThrow('Office-hours design draft: no independent Agent/Task opinion');
expect(() => validateOfficeHoursDesignDraft({ ...draft, toolCalls: [...draft.toolCalls].reverse() }))
.toThrow('no independent Agent/Task opinion');
expect(() => validateOfficeHoursDesignDraft({ ...draft, designContent: draftDesign.replace(/## Success Criteria\n[^\n]+\n/, '') }))
.toThrow('repo design lacks substantive Success Criteria');
expect(() => validateOfficeHoursDesignDraft({ ...draft, designContent: null })).toThrow('repo design is missing');
expect(() => validateOfficeHoursCompletion({ ...completed(), toolCalls: completed().toolCalls.slice(1) }))
.toThrow('Office-hours completion: no independent Agent/Task opinion');
});
test('accepts a completed approved design with unresolved reviewer concerns', () => {
const review = validateOfficeHoursCompletion(completed());
expect(review?.report).toBe(report);
@@ -695,6 +710,7 @@ describe('office-hours completion eval selection', () => {
test('office-hours source selects its dedicated workflow instead of the generic carve file', () => {
const { selected } = selectTests(['office-hours/sections/design-and-handoff.md.tmpl'], E2E_TOUCHFILES);
expect(selected).toContain('office-hours-section-loading');
expect(selected).toContain('office-hours-design-draft');
expect(selected).not.toContain('carve-section-loading');
});
});
+1 -1
View File
@@ -320,7 +320,7 @@ test('new scope route rejects stale, foreign, premature or unsuccessful evidence
p => { p.tools[1]!.toolUseId = 'unrelated'; },
p => { p.tools[1]!.sessionId = 'foreign'; },
p => { p.tools.splice(1, 1); },
p => { p.tools.push(structuredClone(p.tools[1]!)); },
p => { p.tools.push(structuredClone(p.tools[1]!) as never); },
p => { p.tools[2]!.timestamp = new Date(p.opts.commandStartedAt + 1).toISOString(); },
p => { p.tools[2]!.timestamp = new Date(Date.parse(p.tools[1]!.timestamp) - 1).toISOString(); },
p => { p.tools[2]!.timestamp = 'unknown'; },
+1 -1
View File
@@ -5,7 +5,7 @@ const ptyIds = [
'plan-ceo-review-plan-mode', 'plan-eng-review-plan-mode', 'plan-design-review-plan-mode',
'plan-devex-review-plan-mode', 'plan-mode-no-op', 'office-hours-auto-mode',
'auto-decide-preserved', 'plan-ceo-mode-routing', 'plan-design-with-ui-scope', 'plan-eng-finding-floor',
'auq-format-gate', 'carve-section-loading', 'office-hours-section-loading', 'plan-ceo-section-loading', 'ship-section-loading',
'auq-format-gate', 'carve-section-loading', 'office-hours-section-loading', 'office-hours-design-draft', 'plan-ceo-section-loading', 'ship-section-loading',
'plan-ceo-finding-floor', 'plan-design-finding-floor', 'plan-devex-finding-floor',
'plan-eng-multi-finding-batching', 'plan-ceo-split-overflow',
].sort();
@@ -0,0 +1,68 @@
/**
* Office-hours design-draft checkpoint (periodic). The fixed startup interview
* from CARVE_GUARDS['office-hours'] runs only through the Write that creates the
* design (observed at 269s of the full workflow in run 36385945043). It applies
* the full workflow's design-draft checks, required section reads and
* launch/skill-read guards. The complete review, approval and handoff workflow
* is marathon tier in skill-e2e-office-hours-section-loading.test.ts.
*/
import { test, expect } from 'bun:test';
import * as fs from 'node:fs';
import * as path from 'node:path';
import { describeE2ETier } from './helpers/e2e-gate';
import { CAPTURE_LONG_MS } from './helpers/eval-budgets';
import { setupSkillDir, skillFromWorktree, captureSectionReads, LONG_SECTION_CAPTURE_MS } from './helpers/auq-sdk-capture';
import { CARVE_GUARDS } from './helpers/carve-guards';
import { validateOfficeHoursDesignDraft } from './helpers/office-hours-completion';
const describePeriodic = describeE2ETier('periodic');
describePeriodic('/office-hours design-draft checkpoint (periodic)', () => {
test('the startup interview reads its sections and writes a complete design after an independent opinion', async () => {
const guard = CARVE_GUARDS['office-hours'];
const answersStart = guard.scenario.indexOf('This is a synthetic interview fixture');
const answersEnd = guard.scenario.indexOf('Use the available Agent tool');
if (answersStart < 0 || answersEnd < answersStart) {
throw new Error('fixture drift: CARVE_GUARDS office-hours scenario no longer has its interview block');
}
const { skillMd, sectionsFrom } = skillFromWorktree(guard.skill);
const planDir = setupSkillDir({
skillName: guard.skill, skillMd, sectionsFrom, tmpPrefix: 'gstack-office-hours-draft-',
});
const capture = await captureSectionReads({
planDir, skillName: guard.skill,
scenario: `Run office hours in STARTUP mode through the diagnostic, premise challenge,
alternatives, independent opinion and the design document.
${guard.scenario.slice(answersStart, answersEnd).trim()}
Use the available Agent tool for the independent opinion when the workflow calls
for it. It is read-only. An automated test is not a reason to skip it. This fixture
checks a separate real tool call: obtain the independent opinion on RosterCheck
before writing the design. A self-authored "second opinion" is not a subagent result.
Save the design to docs/designs/roster-check.md in this fixture; keep all artifacts
inside the fixture.
Checkpoint for this run: stop after the Write that saves the complete design
document. Do not run the spec review, approval, relationship closing or handoff;
the full workflow covers them separately. Then write REPORT.md with the design path
and a one-line outcome for each phase you completed.`,
testName: 'office-hours-design-draft', runId: `office-hours-design-draft-${process.env.EVALS_RUN_ID ?? 'local'}`,
timeout: LONG_SECTION_CAPTURE_MS,
});
const designPath = path.join(planDir, 'docs/designs/roster-check.md');
expect(capture.exitReason).toBe('success');
validateOfficeHoursDesignDraft({
designPath, toolCalls: capture.toolCalls,
designContent: fs.existsSync(designPath) ? fs.readFileSync(designPath, 'utf-8') : null,
});
const missing = guard.requiredReads.filter(section => !capture.readSections.has(section));
expect({ reportProduced: capture.reportProduced, read: [...capture.readSections], missing }).toEqual({
reportProduced: true, read: expect.any(Array), missing: [],
});
expect(capture.toolCalls.filter(call => call.tool === 'Skill')).toEqual([]);
const ownSkillPath = path.join(planDir, guard.skill, 'SKILL.md');
expect(capture.toolCalls.filter(call => call.tool === 'Read'
&& /(?:^|\/)SKILL\.md$/.test(String(call.input?.file_path ?? ''))
&& path.resolve(planDir, call.input.file_path) !== ownSkillPath)).toEqual([]);
}, CAPTURE_LONG_MS);
});
@@ -1,7 +1,12 @@
/**
* Full office-hours startup workflow, isolated from the generic carve shard.
* Office-hours startup workflow, isolated from the generic carve shard.
* A fixed interview exercises real opinion/design/review/approval/handoff work.
* Free completion regressions live in office-hours-completion.test.ts.
*
* This full start-to-finish workflow (1–3 real spec-review rounds, ~20 min) is
* marathon tier. skill-e2e-office-hours-design-draft.test.ts runs the same
* interview only through the checkpoint that creates the design (~5 min) in
* the periodic lane.
*/
import { test, expect } from 'bun:test';
import * as fs from 'node:fs';
@@ -11,7 +16,7 @@ import { setupSkillDir, skillFromWorktree, captureSectionReads } from './helpers
import { CARVE_GUARDS } from './helpers/carve-guards';
import { validateOfficeHoursCompletion, validateOfficeHoursReviewerHandoffs, validateOfficeHoursReviewArtifacts, validateOfficeHoursReviewPreservation } from './helpers/office-hours-completion';
const describeE2E = describeE2ETier('periodic');
const describeMarathon = describeE2ETier('marathon');
const runId = `office-hours-section-loading-${process.env.EVALS_RUN_ID ?? 'local'}`;
// Full startup diagnosis + outside opinion + up to three spec reviews exceeds
@@ -25,7 +30,7 @@ const runId = `office-hours-section-loading-${process.env.EVALS_RUN_ID ?? 'local
const OFFICE_HOURS_CAPTURE_MS = 1_200_000;
const OFFICE_HOURS_TEST_MS = 1_260_000;
describeE2E('/office-hours full section-loading workflow (periodic)', () => {
describeMarathon('/office-hours full section-loading workflow (marathon)', () => {
test('a real startup review reads its sections and completes the approved design and handoff', async () => {
const guard = CARVE_GUARDS['office-hours'];
const { skillMd, sectionsFrom } = skillFromWorktree(guard.skill);
+1 -1
View File
@@ -510,7 +510,7 @@ describe('TOUCHFILES completeness', () => {
});
test('E2E_TIERS only contains valid tier values', () => {
const validTiers = ['gate', 'periodic'];
const validTiers = ['gate', 'periodic', 'marathon'];
for (const [name, tier] of Object.entries(E2E_TIERS)) {
if (!validTiers.includes(tier)) {
throw new Error(`E2E_TIERS['${name}'] has invalid tier '${tier}'. Valid: ${validTiers.join(', ')}`);