test: move the full office-hours workflow to marathon; add a periodic design-draft checkpoint

The full startup workflow runs 1–3 real spec-review rounds (~280s each) and
hit its 1200s capture in run 36385945043 at finalize. Review depth is the
product's loop, so the case cannot fit a blocking lane without cutting
rounds. It is now marathon tier with every assertion unchanged.

skill-e2e-office-hours-design-draft.test.ts (periodic) runs the same fixed
interview only through the Write that creates the design (269s in that run)
and applies the full validator's design-draft checks, the required section
reads and the launch/foreign-skill-read guards. validateOfficeHoursDesignDraft
is extracted from validateOfficeHoursCompletion, which still applies it.

Selection: office-hours-design-draft is registered periodic; the marathon-only
file is already excluded from the gate and periodic plans by the B5 planner
rule. Tier-alignment regexes and the valid-tier check accept 'marathon'.
A type-only cast in plan-scope-selection.test.ts removes a diagnostic whose
union print order made the ratchet identity unstable; baseline tightened.
This commit is contained in:
garrytan committed 2026-09-29 15:29:46 +00:00
1 parent a41cdb7d7d
commit 4973f96d98
12 files changed
+180 -57

No files matched your search

+44 -27
View File
@@ -91,6 +91,49 @@ function assignmentBody(markdown: string): string {
|| '';
}
/**
* Design-draft phase of the fixed fixture: the repo design carries every
* required section and an Assignment, and an independent Agent/Task opinion on
* RosterCheck preceded the Write that created it. The full workflow validator
* applies these same checks; the focused design-draft capture applies them alone.
*/
export function validateOfficeHoursDesignDraft(
evidence: Pick<OfficeHoursCompletionEvidence, 'designPath' | 'designContent' | 'toolCalls'>,
label = 'Office-hours design draft',
): { designPath: string; repoPath: string; firstDesignWrite: number } {
const fail = (message: string): never => { throw new Error(`${label}: ${message}`); };
if (evidence.designContent === null) fail(`repo design is missing: ${evidence.designPath}`);
const design = evidence.designContent!;
for (const [section, names] of [
['Problem Statement', ['problem statement']],
['Recommended Approach', ['recommended approach']],
['Success Criteria', ['success criteria']],
['What I noticed about how you think', ['what i noticed about how you think']],
] as const) {
if (!substantive(sectionBody(design, [...names]))) fail(`repo design lacks substantive ${section}`);
}
if (!substantive(assignmentBody(design))) fail('repo design lacks a concrete Assignment');
// A cold-read opinion before the design exists is not the required spec
// review. The fixture promises an available Agent, so require an attempt
// that names this design even when the review subsequently fails.
const designPath = evidence.designPath.replace(/\\/g, '/');
const repoPath = designPath.match(/(?:^|\/)(docs\/designs\/[^/]+\.md)$/)?.[1] ?? designPath;
const firstDesignWrite = evidence.toolCalls.findIndex(call => {
const writtenPath = String(call.input?.file_path ?? '').replace(/\\/g, '/').replace(/^\.\//, '');
return call.tool === 'Write' && (writtenPath === designPath || writtenPath === repoPath);
});
if (firstDesignWrite === -1) fail('no observed Write created the repo design');
const opinion = evidence.toolCalls.slice(0, firstDesignWrite).some(call => {
if (!['Agent', 'Task'].includes(call.tool)) return false;
const prompt = `${String(call.input?.description ?? '')}\n${String(call.input?.prompt ?? '')}`;
return /\bRosterCheck\b/i.test(prompt)
&& /\b(?:review|challenge|opinion|critique|perspective|steelman|advisor)\b|\bcold.read\b/i.test(prompt);
});
if (!opinion) fail('no independent Agent/Task opinion on RosterCheck preceded the repo design Write');
return { designPath, repoPath, firstDesignWrite };
}
export function validateOfficeHoursCompletion(evidence: OfficeHoursCompletionEvidence): OfficeHoursReviewEvidence | null {
const fail = (message: string): never => { throw new Error(`Office-hours completion: ${message}`); };
if (evidence.exitReason !== 'success') fail(`execution failed: ${evidence.exitReason}`);
@@ -111,34 +154,8 @@ export function validateOfficeHoursCompletion(evidence: OfficeHoursCompletionEvi
if (statuses.length !== 1 || statuses[0].trim().toUpperCase() !== 'APPROVED') {
fail('repo design is not marked Status: APPROVED');
}
for (const [label, names] of [
['Problem Statement', ['problem statement']],
['Recommended Approach', ['recommended approach']],
['Success Criteria', ['success criteria']],
['What I noticed about how you think', ['what i noticed about how you think']],
] as const) {
if (!substantive(sectionBody(design, [...names]))) fail(`repo design lacks substantive ${label}`);
}
if (!substantive(assignmentBody(design))) fail('repo design lacks a concrete Assignment');
const { designPath, repoPath, firstDesignWrite } = validateOfficeHoursDesignDraft(evidence, 'Office-hours completion');
if (!substantive(assignmentBody(evidence.output))) fail('REPORT.md lacks the Assignment');
// A cold-read opinion before the design exists is not the required spec
// review. The fixture promises an available Agent, so require an attempt
// that names this design even when the review subsequently fails.
const designPath = evidence.designPath.replace(/\\/g, '/');
const repoPath = designPath.match(/(?:^|\/)(docs\/designs\/[^/]+\.md)$/)?.[1] ?? designPath;
const firstDesignWrite = evidence.toolCalls.findIndex(call => {
const writtenPath = String(call.input?.file_path ?? '').replace(/\\/g, '/').replace(/^\.\//, '');
return call.tool === 'Write' && (writtenPath === designPath || writtenPath === repoPath);
});
if (firstDesignWrite === -1) fail('no observed Write created the repo design');
const opinion = evidence.toolCalls.slice(0, firstDesignWrite).some(call => {
if (!['Agent', 'Task'].includes(call.tool)) return false;
const prompt = `${String(call.input?.description ?? '')}\n${String(call.input?.prompt ?? '')}`;
return /\bRosterCheck\b/i.test(prompt)
&& /\b(?:review|challenge|opinion|critique|perspective|steelman|advisor)\b|\bcold.read\b/i.test(prompt);
});
if (!opinion) fail('no independent Agent/Task opinion on RosterCheck preceded the repo design Write');
const reviews = evidence.toolCalls.slice(firstDesignWrite + 1).filter(call => {
if (!['Agent', 'Task'].includes(call.tool)) return false;
const prompt = String(call.input?.prompt ?? '').replace(/\\/g, '/');
+3 -1
View File
@@ -1093,6 +1093,7 @@ export const E2E_TOUCHFILES: Record<string, string[]> = {
'ship/SKILL.md', 'test/helpers/e2e-gate.ts'],
'office-hours-section-loading': [ 'office-hours/**', 'bin/gstack-office-hours-review', 'lib/office-hours-review.ts', 'lib/fs-atomic.ts', 'scripts/resolvers/review.ts', 'scripts/resolvers/sections.ts', 'scripts/gen-skill-docs.ts', 'test/helpers/carve-guards.ts', 'test/helpers/auq-sdk-capture.ts', 'test/helpers/office-hours-completion.ts', 'test/helpers/llm-judge.ts', 'test/helpers/session-runner.ts', 'test/skill-e2e-office-hours-section-loading.test.ts', 'test/helpers/agent-sdk-runner.ts', 'test/helpers/auq-native-capture.ts', 'test/helpers/auto-decision-state.ts', 'test/helpers/autoplan-artifact-digest.ts', 'test/helpers/autoplan-artifact-permission.ts', 'test/helpers/autoplan-artifact-recorder.ts', 'test/helpers/capture-parity-baseline.ts', 'test/helpers/claude-pty-runner.ts', 'test/helpers/dx-selected-navigation.ts', 'test/helpers/e2e-gate.ts', 'test/helpers/eng-cache-writer-decision.ts', 'test/helpers/hermetic-skill-runtime.ts', 'test/helpers/native-auto-decide.ts', 'test/helpers/owned-claude-transcript.ts', 'test/helpers/parity-harness.ts', 'test/helpers/plan-count-artifacts.ts', 'test/helpers/plan-count-file-permission.ts', 'test/helpers/plan-count-fixture.ts', 'test/helpers/plan-count-pending-exit.ts', 'test/helpers/plan-count-pending-question.ts', 'test/helpers/plan-count-transcript.ts', 'test/helpers/plan-floor-review.ts', 'test/helpers/plan-floor-target.ts', 'test/helpers/plan-scope-selection.ts', 'test/helpers/plan-seed-submission.ts', 'test/helpers/plan-skill-question-events.ts', 'test/helpers/plan-skill-question-hook-scope.ts', 'test/helpers/plan-skill-questions.ts', 'test/helpers/pty-screen.ts', 'test/helpers/pty-trust-dialog.ts', 'test/helpers/skill-census.ts'],
'office-hours-design-draft': [ 'office-hours/**', 'lib/office-hours-review.ts', 'scripts/resolvers/review.ts', 'scripts/resolvers/sections.ts', 'scripts/gen-skill-docs.ts', 'test/helpers/carve-guards.ts', 'test/helpers/auq-sdk-capture.ts', 'test/helpers/office-hours-completion.ts', 'test/helpers/llm-judge.ts', 'test/helpers/session-runner.ts', 'test/skill-e2e-office-hours-design-draft.test.ts', 'test/helpers/agent-sdk-runner.ts', 'test/helpers/auq-native-capture.ts', 'test/helpers/auto-decision-state.ts', 'test/helpers/autoplan-artifact-digest.ts', 'test/helpers/autoplan-artifact-permission.ts', 'test/helpers/autoplan-artifact-recorder.ts', 'test/helpers/capture-parity-baseline.ts', 'test/helpers/claude-pty-runner.ts', 'test/helpers/dx-selected-navigation.ts', 'test/helpers/e2e-gate.ts', 'test/helpers/eng-cache-writer-decision.ts', 'test/helpers/hermetic-skill-runtime.ts', 'test/helpers/native-auto-decide.ts', 'test/helpers/owned-claude-transcript.ts', 'test/helpers/parity-harness.ts', 'test/helpers/plan-count-artifacts.ts', 'test/helpers/plan-count-file-permission.ts', 'test/helpers/plan-count-fixture.ts', 'test/helpers/plan-count-pending-exit.ts', 'test/helpers/plan-count-pending-question.ts', 'test/helpers/plan-count-transcript.ts', 'test/helpers/plan-floor-review.ts', 'test/helpers/plan-floor-target.ts', 'test/helpers/plan-scope-selection.ts', 'test/helpers/plan-seed-submission.ts', 'test/helpers/plan-skill-question-events.ts', 'test/helpers/plan-skill-question-hook-scope.ts', 'test/helpers/plan-skill-questions.ts', 'test/helpers/pty-screen.ts', 'test/helpers/pty-trust-dialog.ts', 'test/helpers/skill-census.ts'],
'plan-devex-peer-comparison-classification': [
@@ -1474,7 +1475,8 @@ export const E2E_TIERS: Record<string, 'gate' | 'periodic' | 'marathon'> = {
'arm-benchmark-native-overbuild': 'periodic',
'arm-benchmark-crud-endpoint': 'periodic',
'arm-benchmark-bugfix-decoys': 'periodic',
'office-hours-section-loading': 'periodic', // Full startup design/review/approval workflow
'office-hours-section-loading': 'marathon', // Full startup design/review/approval workflow (1–3 real review rounds, ~20 min)
'office-hours-design-draft': 'periodic', // Same interview through the design-creating Write (~5 min)
'plan-decision-classification': 'periodic',
'plan-devex-peer-comparison-classification': 'periodic',
'health-reporting': 'periodic',