mirror of
https://github.com/garrytan/gstack.git
synced 2026-10-03 01:46:55 +02:00
test: move the full office-hours workflow to marathon; add a periodic design-draft checkpoint
The full startup workflow runs 1–3 real spec-review rounds (~280s each) and hit its 1200s capture in run 36385945043 at finalize. Review depth is the product's loop, so the case cannot fit a blocking lane without cutting rounds. It is now marathon tier with every assertion unchanged. skill-e2e-office-hours-design-draft.test.ts (periodic) runs the same fixed interview only through the Write that creates the design (269s in that run) and applies the full validator's design-draft checks, the required section reads and the launch/foreign-skill-read guards. validateOfficeHoursDesignDraft is extracted from validateOfficeHoursCompletion, which still applies it. Selection: office-hours-design-draft is registered periodic; the marathon-only file is already excluded from the gate and periodic plans by the B5 planner rule. Tier-alignment regexes and the valid-tier check accept 'marathon'. A type-only cast in plan-scope-selection.test.ts removes a diagnostic whose union print order made the ratchet identity unstable; baseline tightened.
This commit is contained in:
1 parent
a41cdb7d7d
commit
4973f96d98
12 files changed
+180
-57
No files matched your search
@@ -2,7 +2,8 @@ import { expect, test } from 'bun:test';
|
||||
import * as fs from 'node:fs';
|
||||
import * as os from 'node:os';
|
||||
import * as path from 'node:path';
|
||||
import { nativePlanCallFingerprint } from './helpers/claude-pty-runner';
|
||||
import { nativePlanCallFingerprint, type AskUserQuestionFingerprint } from './helpers/claude-pty-runner';
|
||||
import type { NativeQuestion } from './helpers/plan-skill-questions';
|
||||
import type { NativePlanQuestionCall, PlanCountTranscript } from './helpers/plan-count-transcript';
|
||||
import { ceoSplitCandidate, ceoSplitDecisionFingerprints, isCeoSplitCandidateCall, isCeoSplitCollectionComplete } from './helpers/ceo-split-question-policy';
|
||||
import captured from './fixtures/ceo-split-collection-0bcd.json';
|
||||
@@ -54,26 +55,28 @@ test.each([0, 1, 2, 3, 4, 5, 6])('the exact original %i-call prefix waits for th
|
||||
// Run 36385945043: the skill cited ledger row IDs ("D2.1 — R-E1: …") and offered
|
||||
// a fourth "Hold, discuss first" option. No candidate was recognized, so collection
|
||||
// never stopped and the attempt ran the whole review (1302s) after the E5 ACK.
|
||||
function rowIdCapture() {
|
||||
const calls = structuredClone(rowIds.calls) as NativePlanQuestionCall[];
|
||||
function rowIdCapture(): { transcript: PlanCountTranscript; fingerprints: AskUserQuestionFingerprint[] } {
|
||||
const calls = structuredClone(rowIds.calls) as unknown as NativePlanQuestionCall[];
|
||||
const transcript: PlanCountTranscript = { status: 'ready', calls, assistantMessages: [] };
|
||||
const fingerprints = rowIds.fingerprints.map((fp, index) => ({ ...structuredClone(fp), nativeCall: calls[index]! }));
|
||||
return { transcript, fingerprints };
|
||||
}
|
||||
const rowIdAccepts = (state: ReturnType<typeof rowIdCapture>) => isCeoSplitCollectionComplete(state.transcript, state.fingerprints);
|
||||
|
||||
test('ledger row-ID candidate questions from run 36385945043 finish collection at the E5 ACK', () => {
|
||||
const state = rowIdCapture();
|
||||
expect(rowIds.provenance.originalOutcome).toBe('completion_summary');
|
||||
expect(rowIds.provenance.originalReviewCount).toBe(0);
|
||||
expect(state.transcript.calls.at(-1)!.answeredAt).toBe(rowIds.provenance.completeAt);
|
||||
expect(state.transcript.calls.map(call => ceoSplitCandidate(call.questions[0]!))).toEqual([null, 'E1', 'E2', 'E3', 'E4', 'E5']);
|
||||
expect(state.transcript.calls.map(call => ceoSplitCandidate(call.questions[0] as NativeQuestion)))
|
||||
.toEqual([null, 'E1', 'E2', 'E3', 'E4', 'E5']);
|
||||
expect(state.fingerprints.map(isCeoSplitCandidateCall)).toEqual([false, true, true, true, true, true]);
|
||||
for (let length = 0; length < 6; length++) {
|
||||
const prefix = rowIdCapture();
|
||||
prefix.transcript.calls.length = length; prefix.fingerprints.length = length;
|
||||
expect(accepts(prefix)).toBe(false);
|
||||
expect(rowIdAccepts(prefix)).toBe(false);
|
||||
}
|
||||
expect(accepts(state)).toBe(true);
|
||||
expect(rowIdAccepts(state)).toBe(true);
|
||||
});
|
||||
|
||||
test.each(['foreign_row', 'quoted_row', 'second_platform', 'held'])('row-ID collection rejects %s evidence', kind => {
|
||||
@@ -84,7 +87,7 @@ test.each(['foreign_row', 'quoted_row', 'second_platform', 'held'])('row-ID coll
|
||||
if (kind === 'second_platform') question.question = question.question.replace('?', ' or the Slack bot?');
|
||||
call.answers = { [question.question]: kind === 'held' ? question.options[3]!.label : selected };
|
||||
state.fingerprints = fromCalls(state.transcript.calls).fingerprints;
|
||||
expect(accepts(state)).toBe(false);
|
||||
expect(rowIdAccepts(state)).toBe(false);
|
||||
});
|
||||
|
||||
test('four candidate calls with five independent tabs meet the original floor', () => {
|
||||
|
||||
@@ -33,13 +33,13 @@ const TEST_DIR = import.meta.dir;
|
||||
// Both quote styles — a mechanical refactor to double quotes must not
|
||||
// silently drop a file from the invariant (fail-open is the defect class
|
||||
// this test exists to kill).
|
||||
const SELF_GATE_RE = /EVALS_TIER\s*===\s*['"](gate|periodic)['"]/g;
|
||||
const SELF_GATE_RE = /EVALS_TIER\s*===\s*['"](gate|periodic|marathon)['"]/g;
|
||||
// Consolidated gate helper (test/helpers/e2e-gate.ts). Both regexes stay
|
||||
// active: migrated files self-gate via `describeE2ETier('<tier>')` (or the
|
||||
// boolean form `e2eTierEnabled('<tier>')`), while stragglers still using the
|
||||
// raw predicate are caught by SELF_GATE_RE above. The tier argument maps to
|
||||
// the declared tier exactly like the raw predicate's tier literal did.
|
||||
const HELPER_GATE_RE = /\b(?:describeE2ETier|e2eTierEnabled)\(\s*['"](gate|periodic)['"]/g;
|
||||
const HELPER_GATE_RE = /\b(?:describeE2ETier|e2eTierEnabled)\(\s*['"](gate|periodic|marathon)['"]/g;
|
||||
|
||||
/**
|
||||
* Ratchet, not amnesty (the contract KNOWN_MATRIX_GAPS pioneered before the
|
||||
|
||||
@@ -91,6 +91,49 @@ function assignmentBody(markdown: string): string {
|
||||
|| '';
|
||||
}
|
||||
|
||||
/**
|
||||
* Design-draft phase of the fixed fixture: the repo design carries every
|
||||
* required section and an Assignment, and an independent Agent/Task opinion on
|
||||
* RosterCheck preceded the Write that created it. The full workflow validator
|
||||
* applies these same checks; the focused design-draft capture applies them alone.
|
||||
*/
|
||||
export function validateOfficeHoursDesignDraft(
|
||||
evidence: Pick<OfficeHoursCompletionEvidence, 'designPath' | 'designContent' | 'toolCalls'>,
|
||||
label = 'Office-hours design draft',
|
||||
): { designPath: string; repoPath: string; firstDesignWrite: number } {
|
||||
const fail = (message: string): never => { throw new Error(`${label}: ${message}`); };
|
||||
if (evidence.designContent === null) fail(`repo design is missing: ${evidence.designPath}`);
|
||||
const design = evidence.designContent!;
|
||||
for (const [section, names] of [
|
||||
['Problem Statement', ['problem statement']],
|
||||
['Recommended Approach', ['recommended approach']],
|
||||
['Success Criteria', ['success criteria']],
|
||||
['What I noticed about how you think', ['what i noticed about how you think']],
|
||||
] as const) {
|
||||
if (!substantive(sectionBody(design, [...names]))) fail(`repo design lacks substantive ${section}`);
|
||||
}
|
||||
if (!substantive(assignmentBody(design))) fail('repo design lacks a concrete Assignment');
|
||||
|
||||
// A cold-read opinion before the design exists is not the required spec
|
||||
// review. The fixture promises an available Agent, so require an attempt
|
||||
// that names this design even when the review subsequently fails.
|
||||
const designPath = evidence.designPath.replace(/\\/g, '/');
|
||||
const repoPath = designPath.match(/(?:^|\/)(docs\/designs\/[^/]+\.md)$/)?.[1] ?? designPath;
|
||||
const firstDesignWrite = evidence.toolCalls.findIndex(call => {
|
||||
const writtenPath = String(call.input?.file_path ?? '').replace(/\\/g, '/').replace(/^\.\//, '');
|
||||
return call.tool === 'Write' && (writtenPath === designPath || writtenPath === repoPath);
|
||||
});
|
||||
if (firstDesignWrite === -1) fail('no observed Write created the repo design');
|
||||
const opinion = evidence.toolCalls.slice(0, firstDesignWrite).some(call => {
|
||||
if (!['Agent', 'Task'].includes(call.tool)) return false;
|
||||
const prompt = `${String(call.input?.description ?? '')}\n${String(call.input?.prompt ?? '')}`;
|
||||
return /\bRosterCheck\b/i.test(prompt)
|
||||
&& /\b(?:review|challenge|opinion|critique|perspective|steelman|advisor)\b|\bcold.read\b/i.test(prompt);
|
||||
});
|
||||
if (!opinion) fail('no independent Agent/Task opinion on RosterCheck preceded the repo design Write');
|
||||
return { designPath, repoPath, firstDesignWrite };
|
||||
}
|
||||
|
||||
export function validateOfficeHoursCompletion(evidence: OfficeHoursCompletionEvidence): OfficeHoursReviewEvidence | null {
|
||||
const fail = (message: string): never => { throw new Error(`Office-hours completion: ${message}`); };
|
||||
if (evidence.exitReason !== 'success') fail(`execution failed: ${evidence.exitReason}`);
|
||||
@@ -111,34 +154,8 @@ export function validateOfficeHoursCompletion(evidence: OfficeHoursCompletionEvi
|
||||
if (statuses.length !== 1 || statuses[0].trim().toUpperCase() !== 'APPROVED') {
|
||||
fail('repo design is not marked Status: APPROVED');
|
||||
}
|
||||
for (const [label, names] of [
|
||||
['Problem Statement', ['problem statement']],
|
||||
['Recommended Approach', ['recommended approach']],
|
||||
['Success Criteria', ['success criteria']],
|
||||
['What I noticed about how you think', ['what i noticed about how you think']],
|
||||
] as const) {
|
||||
if (!substantive(sectionBody(design, [...names]))) fail(`repo design lacks substantive ${label}`);
|
||||
}
|
||||
if (!substantive(assignmentBody(design))) fail('repo design lacks a concrete Assignment');
|
||||
const { designPath, repoPath, firstDesignWrite } = validateOfficeHoursDesignDraft(evidence, 'Office-hours completion');
|
||||
if (!substantive(assignmentBody(evidence.output))) fail('REPORT.md lacks the Assignment');
|
||||
|
||||
// A cold-read opinion before the design exists is not the required spec
|
||||
// review. The fixture promises an available Agent, so require an attempt
|
||||
// that names this design even when the review subsequently fails.
|
||||
const designPath = evidence.designPath.replace(/\\/g, '/');
|
||||
const repoPath = designPath.match(/(?:^|\/)(docs\/designs\/[^/]+\.md)$/)?.[1] ?? designPath;
|
||||
const firstDesignWrite = evidence.toolCalls.findIndex(call => {
|
||||
const writtenPath = String(call.input?.file_path ?? '').replace(/\\/g, '/').replace(/^\.\//, '');
|
||||
return call.tool === 'Write' && (writtenPath === designPath || writtenPath === repoPath);
|
||||
});
|
||||
if (firstDesignWrite === -1) fail('no observed Write created the repo design');
|
||||
const opinion = evidence.toolCalls.slice(0, firstDesignWrite).some(call => {
|
||||
if (!['Agent', 'Task'].includes(call.tool)) return false;
|
||||
const prompt = `${String(call.input?.description ?? '')}\n${String(call.input?.prompt ?? '')}`;
|
||||
return /\bRosterCheck\b/i.test(prompt)
|
||||
&& /\b(?:review|challenge|opinion|critique|perspective|steelman|advisor)\b|\bcold.read\b/i.test(prompt);
|
||||
});
|
||||
if (!opinion) fail('no independent Agent/Task opinion on RosterCheck preceded the repo design Write');
|
||||
const reviews = evidence.toolCalls.slice(firstDesignWrite + 1).filter(call => {
|
||||
if (!['Agent', 'Task'].includes(call.tool)) return false;
|
||||
const prompt = String(call.input?.prompt ?? '').replace(/\\/g, '/');
|
||||
|
||||
@@ -1093,6 +1093,7 @@ export const E2E_TOUCHFILES: Record<string, string[]> = {
|
||||
'ship/SKILL.md', 'test/helpers/e2e-gate.ts'],
|
||||
|
||||
'office-hours-section-loading': [ 'office-hours/**', 'bin/gstack-office-hours-review', 'lib/office-hours-review.ts', 'lib/fs-atomic.ts', 'scripts/resolvers/review.ts', 'scripts/resolvers/sections.ts', 'scripts/gen-skill-docs.ts', 'test/helpers/carve-guards.ts', 'test/helpers/auq-sdk-capture.ts', 'test/helpers/office-hours-completion.ts', 'test/helpers/llm-judge.ts', 'test/helpers/session-runner.ts', 'test/skill-e2e-office-hours-section-loading.test.ts', 'test/helpers/agent-sdk-runner.ts', 'test/helpers/auq-native-capture.ts', 'test/helpers/auto-decision-state.ts', 'test/helpers/autoplan-artifact-digest.ts', 'test/helpers/autoplan-artifact-permission.ts', 'test/helpers/autoplan-artifact-recorder.ts', 'test/helpers/capture-parity-baseline.ts', 'test/helpers/claude-pty-runner.ts', 'test/helpers/dx-selected-navigation.ts', 'test/helpers/e2e-gate.ts', 'test/helpers/eng-cache-writer-decision.ts', 'test/helpers/hermetic-skill-runtime.ts', 'test/helpers/native-auto-decide.ts', 'test/helpers/owned-claude-transcript.ts', 'test/helpers/parity-harness.ts', 'test/helpers/plan-count-artifacts.ts', 'test/helpers/plan-count-file-permission.ts', 'test/helpers/plan-count-fixture.ts', 'test/helpers/plan-count-pending-exit.ts', 'test/helpers/plan-count-pending-question.ts', 'test/helpers/plan-count-transcript.ts', 'test/helpers/plan-floor-review.ts', 'test/helpers/plan-floor-target.ts', 'test/helpers/plan-scope-selection.ts', 'test/helpers/plan-seed-submission.ts', 'test/helpers/plan-skill-question-events.ts', 'test/helpers/plan-skill-question-hook-scope.ts', 'test/helpers/plan-skill-questions.ts', 'test/helpers/pty-screen.ts', 'test/helpers/pty-trust-dialog.ts', 'test/helpers/skill-census.ts'],
|
||||
'office-hours-design-draft': [ 'office-hours/**', 'lib/office-hours-review.ts', 'scripts/resolvers/review.ts', 'scripts/resolvers/sections.ts', 'scripts/gen-skill-docs.ts', 'test/helpers/carve-guards.ts', 'test/helpers/auq-sdk-capture.ts', 'test/helpers/office-hours-completion.ts', 'test/helpers/llm-judge.ts', 'test/helpers/session-runner.ts', 'test/skill-e2e-office-hours-design-draft.test.ts', 'test/helpers/agent-sdk-runner.ts', 'test/helpers/auq-native-capture.ts', 'test/helpers/auto-decision-state.ts', 'test/helpers/autoplan-artifact-digest.ts', 'test/helpers/autoplan-artifact-permission.ts', 'test/helpers/autoplan-artifact-recorder.ts', 'test/helpers/capture-parity-baseline.ts', 'test/helpers/claude-pty-runner.ts', 'test/helpers/dx-selected-navigation.ts', 'test/helpers/e2e-gate.ts', 'test/helpers/eng-cache-writer-decision.ts', 'test/helpers/hermetic-skill-runtime.ts', 'test/helpers/native-auto-decide.ts', 'test/helpers/owned-claude-transcript.ts', 'test/helpers/parity-harness.ts', 'test/helpers/plan-count-artifacts.ts', 'test/helpers/plan-count-file-permission.ts', 'test/helpers/plan-count-fixture.ts', 'test/helpers/plan-count-pending-exit.ts', 'test/helpers/plan-count-pending-question.ts', 'test/helpers/plan-count-transcript.ts', 'test/helpers/plan-floor-review.ts', 'test/helpers/plan-floor-target.ts', 'test/helpers/plan-scope-selection.ts', 'test/helpers/plan-seed-submission.ts', 'test/helpers/plan-skill-question-events.ts', 'test/helpers/plan-skill-question-hook-scope.ts', 'test/helpers/plan-skill-questions.ts', 'test/helpers/pty-screen.ts', 'test/helpers/pty-trust-dialog.ts', 'test/helpers/skill-census.ts'],
|
||||
'plan-devex-peer-comparison-classification': [
|
||||
|
||||
|
||||
@@ -1474,7 +1475,8 @@ export const E2E_TIERS: Record<string, 'gate' | 'periodic' | 'marathon'> = {
|
||||
'arm-benchmark-native-overbuild': 'periodic',
|
||||
'arm-benchmark-crud-endpoint': 'periodic',
|
||||
'arm-benchmark-bugfix-decoys': 'periodic',
|
||||
'office-hours-section-loading': 'periodic', // Full startup design/review/approval workflow
|
||||
'office-hours-section-loading': 'marathon', // Full startup design/review/approval workflow (1–3 real review rounds, ~20 min)
|
||||
'office-hours-design-draft': 'periodic', // Same interview through the design-creating Write (~5 min)
|
||||
'plan-decision-classification': 'periodic',
|
||||
'plan-devex-peer-comparison-classification': 'periodic',
|
||||
'health-reporting': 'periodic',
|
||||
|
||||
@@ -3,25 +3,39 @@ import * as fs from 'node:fs';
|
||||
import * as os from 'node:os';
|
||||
import * as path from 'node:path';
|
||||
import { DEFAULT_SHARD_TIMEOUT_MS } from '../scripts/test-paid-shards';
|
||||
import { CAPTURE_LONG_MS } from './helpers/eval-budgets';
|
||||
|
||||
const casePath = path.join(import.meta.dir, 'skill-e2e-office-hours-section-loading.test.ts');
|
||||
|
||||
// Evaluate the real case registration with inert test/describe functions.
|
||||
// Imports are removed, and the captured paid callback is never invoked.
|
||||
function registeredOptions(): { timeout: number; retry: number } {
|
||||
const source = new Bun.Transpiler({ loader: 'ts' }).transformSync(fs.readFileSync(casePath, 'utf8'))
|
||||
// Evaluate the real case registrations with inert test/describe functions.
|
||||
// Imports are removed, and the captured paid callbacks are never invoked.
|
||||
function registrations(file = casePath): Array<{ tier: string; options: unknown }> {
|
||||
const source = new Bun.Transpiler({ loader: 'ts' }).transformSync(fs.readFileSync(file, 'utf8'))
|
||||
.replace(/^import\b[^;]*;\s*$/gm, '');
|
||||
const registrations: unknown[] = [];
|
||||
new Function('test', 'describeE2ETier', source)(
|
||||
(_name: string, _callback: unknown, options: unknown) => registrations.push(options),
|
||||
() => (_name: string, register: () => void) => register(),
|
||||
const found: Array<{ tier: string; options: unknown }> = [];
|
||||
new Function('test', 'describeE2ETier', 'CAPTURE_LONG_MS', source)(
|
||||
(_name: string, _callback: unknown, options: unknown) => found.at(-1)!.options = options,
|
||||
(tier: string) => (_name: string, register: () => void) => { found.push({ tier, options: undefined }); register(); },
|
||||
CAPTURE_LONG_MS,
|
||||
);
|
||||
expect(registrations).toHaveLength(1);
|
||||
return registrations[0] as { timeout: number; retry: number };
|
||||
return found;
|
||||
}
|
||||
|
||||
test('the full office-hours workflow is one marathon-tier registration', () => {
|
||||
expect(registrations()).toEqual([{ tier: 'marathon', options: { timeout: 1_260_000, retry: 0 } }]);
|
||||
});
|
||||
|
||||
test('the design-draft checkpoint is one periodic case inside the ordinary long capture budget', () => {
|
||||
const draft = path.join(import.meta.dir, 'skill-e2e-office-hours-design-draft.test.ts');
|
||||
expect(registrations(draft)).toEqual([{ tier: 'periodic', options: CAPTURE_LONG_MS }]);
|
||||
const source = fs.readFileSync(draft, 'utf8');
|
||||
expect(source).toContain('timeout: LONG_SECTION_CAPTURE_MS');
|
||||
expect(source).toContain('stop after the Write that saves the complete design');
|
||||
expect(source).toContain('Do not run the spec review, approval, relationship closing or handoff');
|
||||
});
|
||||
|
||||
test('office-hours has one bounded attempt even under the paid runner CLI retry default', () => {
|
||||
const options = registeredOptions();
|
||||
const options = registrations()[0]!.options as { timeout: number; retry: number };
|
||||
expect(options).toEqual({ timeout: 1_260_000, retry: 0 });
|
||||
expect(options.timeout + 120_000).toBeLessThanOrEqual(DEFAULT_SHARD_TIMEOUT_MS);
|
||||
|
||||
|
||||
@@ -3,7 +3,7 @@ import { describe, expect, test } from 'bun:test';
|
||||
import * as fs from 'node:fs';
|
||||
import * as path from 'node:path';
|
||||
import * as os from 'node:os';
|
||||
import { validateOfficeHoursCompletion, validateOfficeHoursReviewerHandoffs, validateOfficeHoursReviewArtifacts, validateOfficeHoursReviewPreservation, validateOfficeHoursSpecSummary, type OfficeHoursCompletionEvidence } from './helpers/office-hours-completion';
|
||||
import { validateOfficeHoursCompletion, validateOfficeHoursDesignDraft, validateOfficeHoursReviewerHandoffs, validateOfficeHoursReviewArtifacts, validateOfficeHoursReviewPreservation, validateOfficeHoursSpecSummary, type OfficeHoursCompletionEvidence } from './helpers/office-hours-completion';
|
||||
import { E2E_TOUCHFILES } from './helpers/touchfiles-data';
|
||||
import { selectTests } from './helpers/test-selection';
|
||||
|
||||
@@ -72,6 +72,21 @@ describe('office-hours fixture completion', () => {
|
||||
expect(instructions).toContain('A failed command remains a failure');
|
||||
});
|
||||
|
||||
test('the design-draft checkpoint applies the full validator\'s design and opinion checks alone', () => {
|
||||
const draftDesign = design.replace('Status: APPROVED', 'Status: DRAFT').replace(/## Reviewer Concerns[\s\S]*$/, '');
|
||||
const draft = { designPath, designContent: draftDesign, toolCalls: completed().toolCalls.slice(0, 2) };
|
||||
expect(validateOfficeHoursDesignDraft(draft)).toEqual({ designPath, repoPath: 'docs/designs/roster-check.md', firstDesignWrite: 1 });
|
||||
expect(() => validateOfficeHoursDesignDraft({ ...draft, toolCalls: draft.toolCalls.slice(1) }))
|
||||
.toThrow('Office-hours design draft: no independent Agent/Task opinion');
|
||||
expect(() => validateOfficeHoursDesignDraft({ ...draft, toolCalls: [...draft.toolCalls].reverse() }))
|
||||
.toThrow('no independent Agent/Task opinion');
|
||||
expect(() => validateOfficeHoursDesignDraft({ ...draft, designContent: draftDesign.replace(/## Success Criteria\n[^\n]+\n/, '') }))
|
||||
.toThrow('repo design lacks substantive Success Criteria');
|
||||
expect(() => validateOfficeHoursDesignDraft({ ...draft, designContent: null })).toThrow('repo design is missing');
|
||||
expect(() => validateOfficeHoursCompletion({ ...completed(), toolCalls: completed().toolCalls.slice(1) }))
|
||||
.toThrow('Office-hours completion: no independent Agent/Task opinion');
|
||||
});
|
||||
|
||||
test('accepts a completed approved design with unresolved reviewer concerns', () => {
|
||||
const review = validateOfficeHoursCompletion(completed());
|
||||
expect(review?.report).toBe(report);
|
||||
@@ -695,6 +710,7 @@ describe('office-hours completion eval selection', () => {
|
||||
test('office-hours source selects its dedicated workflow instead of the generic carve file', () => {
|
||||
const { selected } = selectTests(['office-hours/sections/design-and-handoff.md.tmpl'], E2E_TOUCHFILES);
|
||||
expect(selected).toContain('office-hours-section-loading');
|
||||
expect(selected).toContain('office-hours-design-draft');
|
||||
expect(selected).not.toContain('carve-section-loading');
|
||||
});
|
||||
});
|
||||
|
||||
@@ -320,7 +320,7 @@ test('new scope route rejects stale, foreign, premature or unsuccessful evidence
|
||||
p => { p.tools[1]!.toolUseId = 'unrelated'; },
|
||||
p => { p.tools[1]!.sessionId = 'foreign'; },
|
||||
p => { p.tools.splice(1, 1); },
|
||||
p => { p.tools.push(structuredClone(p.tools[1]!)); },
|
||||
p => { p.tools.push(structuredClone(p.tools[1]!) as never); },
|
||||
p => { p.tools[2]!.timestamp = new Date(p.opts.commandStartedAt + 1).toISOString(); },
|
||||
p => { p.tools[2]!.timestamp = new Date(Date.parse(p.tools[1]!.timestamp) - 1).toISOString(); },
|
||||
p => { p.tools[2]!.timestamp = 'unknown'; },
|
||||
|
||||
@@ -5,7 +5,7 @@ const ptyIds = [
|
||||
'plan-ceo-review-plan-mode', 'plan-eng-review-plan-mode', 'plan-design-review-plan-mode',
|
||||
'plan-devex-review-plan-mode', 'plan-mode-no-op', 'office-hours-auto-mode',
|
||||
'auto-decide-preserved', 'plan-ceo-mode-routing', 'plan-design-with-ui-scope', 'plan-eng-finding-floor',
|
||||
'auq-format-gate', 'carve-section-loading', 'office-hours-section-loading', 'plan-ceo-section-loading', 'ship-section-loading',
|
||||
'auq-format-gate', 'carve-section-loading', 'office-hours-section-loading', 'office-hours-design-draft', 'plan-ceo-section-loading', 'ship-section-loading',
|
||||
'plan-ceo-finding-floor', 'plan-design-finding-floor', 'plan-devex-finding-floor',
|
||||
'plan-eng-multi-finding-batching', 'plan-ceo-split-overflow',
|
||||
].sort();
|
||||
|
||||
@@ -0,0 +1,68 @@
|
||||
/**
|
||||
* Office-hours design-draft checkpoint (periodic). The fixed startup interview
|
||||
* from CARVE_GUARDS['office-hours'] runs only through the Write that creates the
|
||||
* design (observed at 269s of the full workflow in run 36385945043). It applies
|
||||
* the full workflow's design-draft checks, required section reads and
|
||||
* launch/skill-read guards. The complete review, approval and handoff workflow
|
||||
* is marathon tier in skill-e2e-office-hours-section-loading.test.ts.
|
||||
*/
|
||||
import { test, expect } from 'bun:test';
|
||||
import * as fs from 'node:fs';
|
||||
import * as path from 'node:path';
|
||||
import { describeE2ETier } from './helpers/e2e-gate';
|
||||
import { CAPTURE_LONG_MS } from './helpers/eval-budgets';
|
||||
import { setupSkillDir, skillFromWorktree, captureSectionReads, LONG_SECTION_CAPTURE_MS } from './helpers/auq-sdk-capture';
|
||||
import { CARVE_GUARDS } from './helpers/carve-guards';
|
||||
import { validateOfficeHoursDesignDraft } from './helpers/office-hours-completion';
|
||||
|
||||
const describePeriodic = describeE2ETier('periodic');
|
||||
|
||||
describePeriodic('/office-hours design-draft checkpoint (periodic)', () => {
|
||||
test('the startup interview reads its sections and writes a complete design after an independent opinion', async () => {
|
||||
const guard = CARVE_GUARDS['office-hours'];
|
||||
const answersStart = guard.scenario.indexOf('This is a synthetic interview fixture');
|
||||
const answersEnd = guard.scenario.indexOf('Use the available Agent tool');
|
||||
if (answersStart < 0 || answersEnd < answersStart) {
|
||||
throw new Error('fixture drift: CARVE_GUARDS office-hours scenario no longer has its interview block');
|
||||
}
|
||||
const { skillMd, sectionsFrom } = skillFromWorktree(guard.skill);
|
||||
const planDir = setupSkillDir({
|
||||
skillName: guard.skill, skillMd, sectionsFrom, tmpPrefix: 'gstack-office-hours-draft-',
|
||||
});
|
||||
const capture = await captureSectionReads({
|
||||
planDir, skillName: guard.skill,
|
||||
scenario: `Run office hours in STARTUP mode through the diagnostic, premise challenge,
|
||||
alternatives, independent opinion and the design document.
|
||||
${guard.scenario.slice(answersStart, answersEnd).trim()}
|
||||
|
||||
Use the available Agent tool for the independent opinion when the workflow calls
|
||||
for it. It is read-only. An automated test is not a reason to skip it. This fixture
|
||||
checks a separate real tool call: obtain the independent opinion on RosterCheck
|
||||
before writing the design. A self-authored "second opinion" is not a subagent result.
|
||||
Save the design to docs/designs/roster-check.md in this fixture; keep all artifacts
|
||||
inside the fixture.
|
||||
|
||||
Checkpoint for this run: stop after the Write that saves the complete design
|
||||
document. Do not run the spec review, approval, relationship closing or handoff;
|
||||
the full workflow covers them separately. Then write REPORT.md with the design path
|
||||
and a one-line outcome for each phase you completed.`,
|
||||
testName: 'office-hours-design-draft', runId: `office-hours-design-draft-${process.env.EVALS_RUN_ID ?? 'local'}`,
|
||||
timeout: LONG_SECTION_CAPTURE_MS,
|
||||
});
|
||||
const designPath = path.join(planDir, 'docs/designs/roster-check.md');
|
||||
expect(capture.exitReason).toBe('success');
|
||||
validateOfficeHoursDesignDraft({
|
||||
designPath, toolCalls: capture.toolCalls,
|
||||
designContent: fs.existsSync(designPath) ? fs.readFileSync(designPath, 'utf-8') : null,
|
||||
});
|
||||
const missing = guard.requiredReads.filter(section => !capture.readSections.has(section));
|
||||
expect({ reportProduced: capture.reportProduced, read: [...capture.readSections], missing }).toEqual({
|
||||
reportProduced: true, read: expect.any(Array), missing: [],
|
||||
});
|
||||
expect(capture.toolCalls.filter(call => call.tool === 'Skill')).toEqual([]);
|
||||
const ownSkillPath = path.join(planDir, guard.skill, 'SKILL.md');
|
||||
expect(capture.toolCalls.filter(call => call.tool === 'Read'
|
||||
&& /(?:^|\/)SKILL\.md$/.test(String(call.input?.file_path ?? ''))
|
||||
&& path.resolve(planDir, call.input.file_path) !== ownSkillPath)).toEqual([]);
|
||||
}, CAPTURE_LONG_MS);
|
||||
});
|
||||
@@ -1,7 +1,12 @@
|
||||
/**
|
||||
* Full office-hours startup workflow, isolated from the generic carve shard.
|
||||
* Office-hours startup workflow, isolated from the generic carve shard.
|
||||
* A fixed interview exercises real opinion/design/review/approval/handoff work.
|
||||
* Free completion regressions live in office-hours-completion.test.ts.
|
||||
*
|
||||
* This full start-to-finish workflow (1–3 real spec-review rounds, ~20 min) is
|
||||
* marathon tier. skill-e2e-office-hours-design-draft.test.ts runs the same
|
||||
* interview only through the checkpoint that creates the design (~5 min) in
|
||||
* the periodic lane.
|
||||
*/
|
||||
import { test, expect } from 'bun:test';
|
||||
import * as fs from 'node:fs';
|
||||
@@ -11,7 +16,7 @@ import { setupSkillDir, skillFromWorktree, captureSectionReads } from './helpers
|
||||
import { CARVE_GUARDS } from './helpers/carve-guards';
|
||||
import { validateOfficeHoursCompletion, validateOfficeHoursReviewerHandoffs, validateOfficeHoursReviewArtifacts, validateOfficeHoursReviewPreservation } from './helpers/office-hours-completion';
|
||||
|
||||
const describeE2E = describeE2ETier('periodic');
|
||||
const describeMarathon = describeE2ETier('marathon');
|
||||
const runId = `office-hours-section-loading-${process.env.EVALS_RUN_ID ?? 'local'}`;
|
||||
|
||||
// Full startup diagnosis + outside opinion + up to three spec reviews exceeds
|
||||
@@ -25,7 +30,7 @@ const runId = `office-hours-section-loading-${process.env.EVALS_RUN_ID ?? 'local
|
||||
const OFFICE_HOURS_CAPTURE_MS = 1_200_000;
|
||||
const OFFICE_HOURS_TEST_MS = 1_260_000;
|
||||
|
||||
describeE2E('/office-hours full section-loading workflow (periodic)', () => {
|
||||
describeMarathon('/office-hours full section-loading workflow (marathon)', () => {
|
||||
test('a real startup review reads its sections and completes the approved design and handoff', async () => {
|
||||
const guard = CARVE_GUARDS['office-hours'];
|
||||
const { skillMd, sectionsFrom } = skillFromWorktree(guard.skill);
|
||||
|
||||
@@ -510,7 +510,7 @@ describe('TOUCHFILES completeness', () => {
|
||||
});
|
||||
|
||||
test('E2E_TIERS only contains valid tier values', () => {
|
||||
const validTiers = ['gate', 'periodic'];
|
||||
const validTiers = ['gate', 'periodic', 'marathon'];
|
||||
for (const [name, tier] of Object.entries(E2E_TIERS)) {
|
||||
if (!validTiers.includes(tier)) {
|
||||
throw new Error(`E2E_TIERS['${name}'] has invalid tier '${tier}'. Valid: ${validTiers.join(', ')}`);
|
||||
|
||||
Reference in new issue
Block a user