test(qa-callers): deterministic child transport, completion-time handoff reads, compact phase report

The exploratory caller cases exist to prove the caller starts and bounds
exploratory QA. Their native adversarial reviewer (review) and plan audit
(ship plan-checks) now come from recorded child outputs instead of a live
subagent, handoff freshness reads are required before completion records
rather than every bookkeeping log, and the phase report is compact. Measured:
194-257 s per case against 208-284 s before, no subagent calls.
This commit is contained in:
garrytan committed 2026-09-30 16:19:39 +00:00
1 parent 3e5ec6df17
commit dd5708e6e8
2 files changed
+17 -4

No files matched your search

+16 -3
View File
@@ -477,6 +477,7 @@ export interface QaCallerFixture {
reviewStart?: string; reviewStart?: string;
gitEnvironment: Record<'GIT_OBJECT_DIRECTORY' | 'GIT_ALTERNATE_OBJECT_DIRECTORIES', string>; gitEnvironment: Record<'GIT_OBJECT_DIRECTORY' | 'GIT_ALTERNATE_OBJECT_DIRECTORIES', string>;
journal: string; journal: string;
child?: string;
mutationEvents: string[]; mutationEvents: string[];
observerErrors: string[]; observerErrors: string[];
observation?: QAWriteObservation; observation?: QAWriteObservation;
@@ -542,6 +543,18 @@ process.exit(exit ?? 127);
if (caseId === 'ship-exploratory-plan-checks') { if (caseId === 'ship-exploratory-plan-checks') {
fs.writeFileSync(path.join(cwd, 'PLAN.md'), '# Scale change\n\n## Verification\n\nThe upper boundary is a required release check: run `bun scripts/probe.ts 9`; require exit 0, stdout `18\\n`, and empty stderr. Ordinary positive input is not a substitute for this check.\n'); fs.writeFileSync(path.join(cwd, 'PLAN.md'), '# Scale change\n\n## Verification\n\nThe upper boundary is a required release check: run `bun scripts/probe.ts 9`; require exit 0, stdout `18\\n`, and empty stderr. Ordinary positive input is not a substitute for this check.\n');
} }
const childOutputs: Partial<Record<QaCallerCase, string>> = {
'review-exploratory-small-cli': `ADVERSARIAL REVIEW (Claude subagent):
Fixtures and tests were reviewed in summary mode; none changed.
1. FIXABLE — scale.ts:3: the added \`!n\` guard treats the documented valid input 0 as invalid because Number("0") is falsy, so \`0\` now exits 2 with the rejection message instead of printing 0.
Recommendation: Fix the \`!n\` guard at scale.ts:3 because it rejects the documented lower boundary 0.
`,
'ship-exploratory-plan-checks': `{"total_items":0,"done":0,"changed":0,"partial":0,"not_done":0,"unverifiable":0,"summary":"PLAN.md has no implementation deliverables. Execution-only check retained verbatim for Step 8.1/9: run \`bun scripts/probe.ts 9\`; require exit 0, stdout \`18\\\\n\`, and empty stderr (PLAN.md, Verification). Pending execution."}
`,
};
const childText = childOutputs[caseId];
const child = childText === undefined ? undefined : path.join(root, 'children', caller === 'review' ? 'adversarial.md' : 'plan-audit.json');
if (child) { fs.mkdirSync(path.dirname(child)); fs.writeFileSync(child, childText!); }
const run = (...args: string[]) => { const run = (...args: string[]) => {
const result = spawnSync('git', args, { cwd, encoding: 'utf8', timeout: 5000 }); const result = spawnSync('git', args, { cwd, encoding: 'utf8', timeout: 5000 });
if (result.status !== 0) throw new Error(`Caller fixture git ${args[0]} failed: ${result.stderr || result.error?.message}`); if (result.status !== 0) throw new Error(`Caller fixture git ${args[0]} failed: ${result.stderr || result.error?.message}`);
@@ -608,7 +621,7 @@ process.exit(exit ?? 127);
const snapshot = () => callerSnapshot({ ...Object.fromEntries(productFiles.map(file => [file, fs.readFileSync(path.join(cwd, file), 'utf8')])), 'fixture.json': fs.readFileSync(fixtureInput, 'utf8') }); const snapshot = () => callerSnapshot({ ...Object.fromEntries(productFiles.map(file => [file, fs.readFileSync(path.join(cwd, file), 'utf8')])), 'fixture.json': fs.readFileSync(fixtureInput, 'utf8') });
const probes = () => fs.readFileSync(journal, 'utf8').split('\n').filter(Boolean).map(line => JSON.parse(line) as CallerProbe); const probes = () => fs.readFileSync(journal, 'utf8').split('\n').filter(Boolean).map(line => JSON.parse(line) as CallerProbe);
const fixture: QaCallerFixture = { const fixture: QaCallerFixture = {
root, cwd, state, runtime, config, caller, caseId, instructions, journal, mutationEvents, observerErrors, workflowCommands, reviewStart, gitEnvironment, root, cwd, state, runtime, config, caller, caseId, instructions, journal, child, mutationEvents, observerErrors, workflowCommands, reviewStart, gitEnvironment,
lateApplied: false, snapshot, probes, lateApplied: false, snapshot, probes,
observe: async () => { observe: async () => {
if (observer || fixture.observation) throw new Error('Caller observation cannot restart mid-capture'); if (observer || fixture.observation) throw new Error('Caller observation cannot restart mid-capture');
@@ -649,10 +662,10 @@ export function callerReviewRecordTemplate(fixture: Pick<QaCallerFixture, 'calle
export function qaCallerSessionOptions(fixture: QaCallerFixture, runId: string): Parameters<typeof runSkillTest>[0] { export function qaCallerSessionOptions(fixture: QaCallerFixture, runId: string): Parameters<typeof runSkillTest>[0] {
return { return {
prompt: `Load gstack's /${fixture.caller} supplied parent phase from caller-${fixture.caller}.md and resume it on the selected working-tree diff against origin/main. This excerpt comes from ${fixture.runtime}/${fixture.caller}/SKILL.md; resolve installed-relative references there, not from the excerpt file or product directory. That path identifies the asset base, not another entrypoint: do not read or invoke the full parent SKILL.md or rerun its preamble. Earlier preamble/branch/base setup is complete; use the existing local origin/main ref without fetch. Earlier-phase asset locators are ${fixture.runtime}/review/checklist.md and ${fixture.runtime}/qa/templates/functional-report-template.md. Read those files directly when referenced; recursive Glob does not follow the installed asset symlinks. Cross-project learnings are configured off in this owned fixture. ${fixture.reviewStart ? `The actual review-start helper already captured REVIEW_START=${fixture.reviewStart} for this unchanged core pass; retain that token. ` : ''} This fixture evaluates only the supplied parent phase, not later publication stages. Read README.md for the project contract and commands. Use diagnostic-client commands such as \`bun scripts/probe.ts <literal>\` for exploratory discoveries and their checkpoint evidence. A required \`bun run test\` is separate suite verification: report it as verification, never as a diagnostic observation or checkpoint anchor/target. Use the production evidence helper to publish each diagnostic checkpoint as \`reports/exploration-NNN.json\`, not inside a nested directory; do not transcribe its observed payload. ${fixture.caseId === 'ship-exploratory-plan-checks' ? 'The previously discovered plan is PLAN.md.' : 'No plan file was found.'} There is no remote service and no release publication is authorized. There is no interactive approver; do not invent answers or permission. Keep normal parent decision gates. Before every completion report or bookkeeping log, read HANDOFF.md and reports/HANDOFF.md if present for any concurrent collaborator update, await the results, and compare evidence with current inputs. A gstack-review-log completed:true record is a completion, not preliminary bookkeeping; a later handoff read cannot validate an earlier completion.\n\nDeadline bookkeeping additionally permits \`bun ${fixture.runtime}/bin/gstack-qa-deadline start ${fixture.cwd}/reports/deadline.json SECONDS [EARLIER_UTC]\`, \`bun ${fixture.runtime}/bin/gstack-qa-deadline status ${fixture.cwd}/reports/deadline.json\`, and \`bun ${fixture.runtime}/bin/gstack-qa-deadline run ${fixture.cwd}/reports/deadline.json -- bun scripts/probe.ts [literal]\`. These are closed literal forms: SECONDS must be positive and at most 300, EARLIER_UTC is the optional caller absolute deadline: use the section clock's Hard deadline UTC, never its Runner entry UTC, reserve-start time or a clock-read time. The child is only the existing diagnostic client with zero or one literal argument. Resolve these exact helper and state paths; do not use variables, another helper, another state file, nested wrappers, scripts, operators or substitutions. Only this helper may create or change reports/deadline.json and its .qa-deadline- temporary files; never use Write/Edit/MultiEdit on those paths. Record the full outer run command in checkpoints and evidence; keep the unchanged child JSON as observed, separate from prefixed guard diagnostics. A completed expired guard-run is not a probe or a pass: retain its unused checkpoint, report not-run coverage and do not restart the deadline. Keep the 12-probe smoke limit. Required suites and explicit plan checks are outside the bounded smoke budget, not permission to reset it.\n\nFunctional evidence uses the same production helper and existing diagnostic client: \`bun ${fixture.runtime}/bin/gstack-qa-evidence capture ${fixture.cwd}/reports NNN --public --deadline ${fixture.cwd}/reports/deadline.json -- bun scripts/probe.ts [literal]\`. These diagnostic receipts are declared public/synthetic, so --public is approved; a fresh three-digit ID is required each time. Explicit plan probes outside the smoke budget may replace --deadline with --timeout-ms 10000; this does not reset or bypass the smoke deadline. Publish causal intent with \`bun ${fixture.runtime}/bin/gstack-qa-evidence checkpoint ${fixture.cwd}/reports NNN CAPTURE_ID 'full prior capture command' 'causal hypothesis' 'full next capture command'\`; quote arguments literally. For complex quoting, Write only capture, observationCommand, hypothesis and nextCommand to reports/intent.json; publish with the same helper: checkpoint REPORT_ROOT NNN intent.json. Materialize is supported when evidence.json is required. Sources stay inside reports. Decide to execute the next probe before publishing its checkpoint, then await successful publication and dispatch that exact probe. If you defer an optional idea or stop exploration, do not publish a checkpoint for it; descri Line truncated prompt: `Load gstack's /${fixture.caller} supplied parent phase from caller-${fixture.caller}.md and resume it on the selected working-tree diff against origin/main. This excerpt comes from ${fixture.runtime}/${fixture.caller}/SKILL.md; resolve installed-relative references there, not from the excerpt file or product directory. That path identifies the asset base, not another entrypoint: do not read or invoke the full parent SKILL.md or rerun its preamble. Earlier preamble/branch/base setup is complete; use the existing local origin/main ref without fetch. Earlier-phase asset locators are ${fixture.runtime}/review/checklist.md and ${fixture.runtime}/qa/templates/functional-report-template.md. Read those files directly when referenced; recursive Glob does not follow the installed asset symlinks. Cross-project learnings are configured off in this owned fixture. ${fixture.reviewStart ? `The actual review-start helper already captured REVIEW_START=${fixture.reviewStart} for this unchanged core pass; retain that token. ` : ''} This fixture evaluates only the supplied parent phase, not later publication stages. Read README.md for the project contract and commands. Use diagnostic-client commands such as \`bun scripts/probe.ts <literal>\` for exploratory discoveries and their checkpoint evidence. A required \`bun run test\` is separate suite verification: report it as verification, never as a diagnostic observation or checkpoint anchor/target. Use the production evidence helper to publish each diagnostic checkpoint as \`reports/exploration-NNN.json\`, not inside a nested directory; do not transcribe its observed payload. ${fixture.caseId === 'ship-exploratory-plan-checks' ? 'The previously discovered plan is PLAN.md.' : 'No plan file was found.'} There is no remote service and no release publication is authorized. There is no interactive approver; do not invent answers or permission. Keep normal parent decision gates. Before the completion report and each completed:true review record, read HANDOFF.md and reports/HANDOFF.md if present for any concurrent collaborator update, await the results, and compare evidence with current inputs. A gstack-review-log completed:true record is a completion, not preliminary bookkeeping; a later handoff read cannot validate an earlier completion.\n\nDeadline bookkeeping additionally permits \`bun ${fixture.runtime}/bin/gstack-qa-deadline start ${fixture.cwd}/reports/deadline.json SECONDS [EARLIER_UTC]\`, \`bun ${fixture.runtime}/bin/gstack-qa-deadline status ${fixture.cwd}/reports/deadline.json\`, and \`bun ${fixture.runtime}/bin/gstack-qa-deadline run ${fixture.cwd}/reports/deadline.json -- bun scripts/probe.ts [literal]\`. These are closed literal forms: SECONDS must be positive and at most 300, EARLIER_UTC is the optional caller absolute deadline: use the section clock's Hard deadline UTC, never its Runner entry UTC, reserve-start time or a clock-read time. The child is only the existing diagnostic client with zero or one literal argument. Resolve these exact helper and state paths; do not use variables, another helper, another state file, nested wrappers, scripts, operators or substitutions. Only this helper may create or change reports/deadline.json and its .qa-deadline- temporary files; never use Write/Edit/MultiEdit on those paths. Record the full outer run command in checkpoints and evidence; keep the unchanged child JSON as observed, separate from prefixed guard diagnostics. A completed expired guard-run is not a probe or a pass: retain its unused checkpoint, report not-run coverage and do not restart the deadline. Keep the 12-probe smoke limit. Required suites and explicit plan checks are outside the bounded smoke budget, not permission to reset it.\n\nFunctional evidence uses the same production helper and existing diagnostic client: \`bun ${fixture.runtime}/bin/gstack-qa-evidence capture ${fixture.cwd}/reports NNN --public --deadline ${fixture.cwd}/reports/deadline.json -- bun scripts/probe.ts [literal]\`. These diagnostic receipts are declared public/synthetic, so --public is approved; a fresh three-digit ID is required each time. Explicit plan probes outside the smoke budget may replace --deadline with --timeout-ms 10000; this does not reset or bypass the smoke deadline. Publish causal intent with \`bun ${fixture.runtime}/bin/gstack-qa-evidence checkpoint ${fixture.cwd}/reports NNN CAPTURE_ID 'full prior capture command' 'causal hypothesis' 'full next capture command'\`; quote arguments literally. For complex quoting, Write only capture, observationCommand, hypothesis and nextCommand to reports/intent.json; publish with the same helper: checkpoint REPORT_ROOT NNN intent.json. Materialize is supported when evidence.json is required. Sources stay inside reports. Decide to execute the next probe before publishing its checkpoint, then await successful publication and dispatch that exact probe. If you defer an optional idea or stop exploration, do not publish a checkpoi Line truncated
appendSystemPrompt: `Caller execution scheduling (fixture contract): appendSystemPrompt: `Caller execution scheduling (fixture contract):
This session has at most 25 assistant turns, including required verification and final artifacts. The command boundary applies to each Bash call, not to the number of independent tool calls in an assistant turn. This session has at most 25 assistant turns, including required verification and final artifacts. The command boundary applies to each Bash call, not to the number of independent tool calls in an assistant turn.
After required clock and approval prerequisites settle, issue independent source Reads and read-only discovery together as separate native tool calls once their paths and inputs are known. Wait for their results before decisions that depend on them. After required clock and approval prerequisites settle, issue independent source Reads and read-only discovery together as separate native tool calls once their paths and inputs are known. For example, after the first clock read, one response can Read the caller excerpt, README.md, every product file and the referenced checklist and templates. Wait for their results before decisions that depend on them.
The completion reserve is for required verification, affected-input revalidation and artifacts, not an earlier deadline. Keep completing required work within the actual remaining deadline; reserve entry alone is not a reason to stop. Use each native probe's snapshot to distinguish current from superseded evidence before deciding which checks still need revalidation. The completion reserve is for required verification, affected-input revalidation and artifacts, not an earlier deadline. Keep completing required work within the actual remaining deadline; reserve entry alone is not a reason to stop. Use each native probe's snapshot to distinguish current from superseded evidence before deciding which checks still need revalidation.
Never group diagnostic probes, checkpoint publication with its next probe, or any action with the clock/status/approval result it needs. Shell composition remains forbidden outside the declared forms. Preserve every required Read, probe, verification, freshness check, approval and report field; the turn limit does not authorize skipping work or reporting incomplete work as passed.`, Never group diagnostic probes, checkpoint publication with its next probe, or any action with the clock/status/approval result it needs. Shell composition remains forbidden outside the declared forms. Preserve every required Read, probe, verification, freshness check, approval and report field; the turn limit does not authorize skipping work or reporting incomplete work as passed.`,
workingDirectory: fixture.cwd, workingDirectory: fixture.cwd,
+1 -1
View File
@@ -1365,7 +1365,7 @@ describe('real caller-specific native fixture and capture boundary', () => {
expect(options.appendSystemPrompt).not.toMatch(/bun scripts\/probe\.ts \d|invalid input|highest.risk/i); expect(options.appendSystemPrompt).not.toMatch(/bun scripts\/probe\.ts \d|invalid input|highest.risk/i);
expect(options.prompt).toContain('Keep normal parent decision gates.'); expect(options.prompt).toContain('Keep normal parent decision gates.');
expect(options.prompt).toContain("use the section clock's Hard deadline UTC, never its Runner entry UTC, reserve-start time or a clock-read time"); expect(options.prompt).toContain("use the section clock's Hard deadline UTC, never its Runner entry UTC, reserve-start time or a clock-read time");
expect(options.prompt).toContain('Before every completion report or bookkeeping log, read HANDOFF.md'); expect(options.prompt).toContain('Before the completion report and each completed:true review record, read HANDOFF.md');
expect(options.prompt).toContain('a later handoff read cannot validate an earlier completion'); expect(options.prompt).toContain('a later handoff read cannot validate an earlier completion');
expect(options.prompt).toContain('If you defer an optional idea or stop exploration, do not publish a checkpoint for it'); expect(options.prompt).toContain('If you defer an optional idea or stop exploration, do not publish a checkpoint for it');
expect(options.prompt).toContain('An unused checkpoint requires an actual authenticated expired-capture result; nearing the deadline or choosing to stop is not enough'); expect(options.prompt).toContain('An unused checkpoint requires an actual authenticated expired-capture result; nearing the deadline or choosing to stop is not enough');