test(qa-callers): hand the caller phase its invocation-start observations and review token; fix(next-version): fetch without auto maintenance

- Every caller case receives the diff, status, log, untracked list, HEAD and an
  already-captured review start token, so the phase spends its budget on the
  contract under test instead of re-running setup reads.
- gstack-next-version's fetches pass --no-auto-maintenance. On git 2.55 a
  completed fetch forks detached maintenance in the caller's repository; the
  free suite's live smoke test ran it inside the CI checkout, and every
  shard-12 pre-push hook hang so far followed a completed smoke fetch.
This commit is contained in:
garrytan committed 2026-09-30 17:59:23 +00:00
1 parent a7872aaae0
commit 157a5ff520
3 files changed
+17 -6

No files matched your search

+3 -3
View File
@@ -165,7 +165,7 @@ function zeroBaseAtLocalWidth(versionPath: string, repoRoot: string): string {
function readBaseVersion(base: string, versionPath: string, repoRoot: string, warnings: string[]): string {
// git fetch is best-effort; we tolerate failure and fall back to whatever
// origin/<base> currently points at.
runCommand("git", ["fetch", "origin", base, "--quiet"], 10000);
runCommand("git", ["fetch", "--no-auto-maintenance", "origin", base, "--quiet"], 10000);
const r = runCommand("git", ["show", `origin/${base}:${versionPath}`]);
if (!r.ok) {
const assumed = zeroBaseAtLocalWidth(versionPath, repoRoot);
@@ -610,7 +610,7 @@ function fetchGitClaimed(
// bounded) brings every missing tip local in a single round trip.
spawnSync(
"git",
["fetch", "origin", ...pending.map((p) => `refs/heads/${p.branch}`), "--depth=1", "--no-tags"],
["fetch", "--no-auto-maintenance", "origin", ...pending.map((p) => `refs/heads/${p.branch}`), "--depth=1", "--no-tags"],
{ encoding: "utf8", timeout: 15000, env: { ...process.env, GIT_TERMINAL_PROMPT: "0" } },
);
// One unservable ref (dangling sha on the server) fails the WHOLE batch
@@ -626,7 +626,7 @@ function fetchGitClaimed(
retries++;
spawnSync(
"git",
["fetch", "origin", `refs/heads/${branch}`, "--depth=1", "--no-tags"],
["fetch", "--no-auto-maintenance", "origin", `refs/heads/${branch}`, "--depth=1", "--no-tags"],
{ encoding: "utf8", timeout: 5000, env: { ...process.env, GIT_TERMINAL_PROMPT: "0" } },
);
outcome = readClaim(branch, sha);
+1
View File
@@ -800,6 +800,7 @@ describe("fetchGitClaimed — unfetched live claims (G2: ls-remote advertises SH
.split("\n")
.filter((l) => l.startsWith("fetch "));
expect(fetches.length).toBe(1);
expect(fetches[0]).toContain("--no-auto-maintenance");
for (const v of ["0-1-70-0", "0-1-71-0", "0-1-72-0"]) {
expect(fetches[0]).toContain(`refs/heads/late-${v}`);
}
+13 -3
View File
@@ -475,6 +475,7 @@ export interface QaCallerFixture {
caller: QaCaller;
caseId: QaCallerCase;
reviewStart?: string;
startObservations: string;
gitEnvironment: Record<'GIT_OBJECT_DIRECTORY' | 'GIT_ALTERNATE_OBJECT_DIRECTORIES', string>;
journal: string;
child?: string;
@@ -588,8 +589,17 @@ Recommendation: Fix the \`!n\` guard at scale.ts:3 because it rejects the docume
GIT_ALTERNATE_OBJECT_DIRECTORIES: '',
};
fs.cpSync(fs.realpathSync(path.join(cwd, '.git/objects')), gitEnvironment.GIT_OBJECT_DIRECTORY, { recursive: true });
const startObservations = [
`git rev-parse --short HEAD: ${run('rev-parse', '--short', 'HEAD')}`,
`git log origin/main..HEAD --oneline: ${run('log', 'origin/main..HEAD', '--oneline') || '(no commits)'}`,
`git ls-files --others --exclude-standard: ${run('ls-files', '--others', '--exclude-standard') || '(none)'}`,
`git status --short: ${run('status', '--short')}`,
`git diff origin/main --stat:\n${run('diff', 'origin/main', '--stat')}`,
`git diff origin/main:\n${run('diff', 'origin/main')}`,
'reports/ and .qa-state/ contain no files yet.',
].join('\n');
let reviewStart: string | undefined;
if (caller === 'review') {
{
const start = spawnSync('bash', [path.join(QA_CALLER_ROOT, 'bin/gstack-review-log'), '--start', 'review'], {
cwd, env: { ...process.env, ...gitEnvironment, GSTACK_HOME: state, GSTACK_STATE_ROOT: state },
encoding: 'utf8', timeout: 5000,
@@ -621,7 +631,7 @@ Recommendation: Fix the \`!n\` guard at scale.ts:3 because it rejects the docume
const snapshot = () => callerSnapshot({ ...Object.fromEntries(productFiles.map(file => [file, fs.readFileSync(path.join(cwd, file), 'utf8')])), 'fixture.json': fs.readFileSync(fixtureInput, 'utf8') });
const probes = () => fs.readFileSync(journal, 'utf8').split('\n').filter(Boolean).map(line => JSON.parse(line) as CallerProbe);
const fixture: QaCallerFixture = {
root, cwd, state, runtime, config, caller, caseId, instructions, journal, child, mutationEvents, observerErrors, workflowCommands, reviewStart, gitEnvironment,
root, cwd, state, runtime, config, caller, caseId, instructions, journal, child, mutationEvents, observerErrors, workflowCommands, reviewStart, startObservations, gitEnvironment,
lateApplied: false, snapshot, probes,
observe: async () => {
if (observer || fixture.observation) throw new Error('Caller observation cannot restart mid-capture');
@@ -662,7 +672,7 @@ export function callerReviewRecordTemplate(fixture: Pick<QaCallerFixture, 'calle
export function qaCallerSessionOptions(fixture: QaCallerFixture, runId: string): Parameters<typeof runSkillTest>[0] {
return {
prompt: `Load gstack's /${fixture.caller} supplied parent phase from caller-${fixture.caller}.md and resume it on the selected working-tree diff against origin/main. This excerpt comes from ${fixture.runtime}/${fixture.caller}/SKILL.md; resolve installed-relative references there, not from the excerpt file or product directory. That path identifies the asset base, not another entrypoint: do not read or invoke the full parent SKILL.md or rerun its preamble. Earlier preamble/branch/base setup is complete; use the existing local origin/main ref without fetch. Earlier-phase asset locators are ${fixture.runtime}/review/checklist.md and ${fixture.runtime}/qa/templates/functional-report-template.md. Read those files directly when referenced; recursive Glob does not follow the installed asset symlinks. Cross-project learnings are configured off in this owned fixture. ${fixture.reviewStart ? `The actual review-start helper already captured REVIEW_START=${fixture.reviewStart} for this unchanged core pass; retain that token. ` : ''} This fixture evaluates only the supplied parent phase, not later publication stages. Read README.md for the project contract and commands. Use diagnostic-client commands such as \`bun scripts/probe.ts <literal>\` for exploratory discoveries and their checkpoint evidence. A required \`bun run test\` is separate suite verification: report it as verification, never as a diagnostic observation or checkpoint anchor/target. Use the production evidence helper to publish each diagnostic checkpoint as \`reports/exploration-NNN.json\`, not inside a nested directory; do not transcribe its observed payload. ${fixture.caseId === 'ship-exploratory-plan-checks' ? 'The previously discovered plan is PLAN.md.' : 'No plan file was found.'} There is no remote service and no release publication is authorized. There is no interactive approver; do not invent answers or permission. Keep normal parent decision gates. Before the completion report and each completed:true review record, read HANDOFF.md and reports/HANDOFF.md if present for any concurrent collaborator update, await the results, and compare evidence with current inputs. A gstack-review-log completed:true record is a completion, not preliminary bookkeeping; a later handoff read cannot validate an earlier completion.\n\nDeadline bookkeeping additionally permits \`bun ${fixture.runtime}/bin/gstack-qa-deadline start ${fixture.cwd}/reports/deadline.json SECONDS [EARLIER_UTC]\`, \`bun ${fixture.runtime}/bin/gstack-qa-deadline status ${fixture.cwd}/reports/deadline.json\`, and \`bun ${fixture.runtime}/bin/gstack-qa-deadline run ${fixture.cwd}/reports/deadline.json -- bun scripts/probe.ts [literal]\`. These are closed literal forms: SECONDS must be positive and at most 300, EARLIER_UTC is the optional caller absolute deadline: use the section clock's Hard deadline UTC, never its Runner entry UTC, reserve-start time or a clock-read time. The child is only the existing diagnostic client with zero or one literal argument. Resolve these exact helper and state paths; do not use variables, another helper, another state file, nested wrappers, scripts, operators or substitutions. Only this helper may create or change reports/deadline.json and its .qa-deadline- temporary files; never use Write/Edit/MultiEdit on those paths. Record the full outer run command in checkpoints and evidence; keep the unchanged child JSON as observed, separate from prefixed guard diagnostics. A completed expired guard-run is not a probe or a pass: retain its unused checkpoint, report not-run coverage and do not restart the deadline. Keep the 12-probe smoke limit. Required suites and explicit plan checks are outside the bounded smoke budget, not permission to reset it.\n\nFunctional evidence uses the same production helper and existing diagnostic client: \`bun ${fixture.runtime}/bin/gstack-qa-evidence capture ${fixture.cwd}/reports NNN --public --deadline ${fixture.cwd}/reports/deadline.json -- bun scripts/probe.ts [literal]\`. These diagnostic receipts are declared public/synthetic, so --public is approved; a fresh three-digit ID is required each time. Explicit plan probes outside the smoke budget may replace --deadline with --timeout-ms 10000; this does not reset or bypass the smoke deadline. Publish causal intent with \`bun ${fixture.runtime}/bin/gstack-qa-evidence checkpoint ${fixture.cwd}/reports NNN CAPTURE_ID 'full prior capture command' 'causal hypothesis' 'full next capture command'\`; quote arguments literally. For complex quoting, Write only capture, observationCommand, hypothesis and nextCommand to reports/intent.json; publish with the same helper: checkpoint REPORT_ROOT NNN intent.json. Materialize is supported when evidence.json is required. Sources stay inside reports. Decide to execute the next probe before publishing its checkpoint, then await successful publication and dispatch that exact probe. If you defer an optional idea or stop exploration, do not publish a checkpoi Line truncated
prompt: `Load gstack's /${fixture.caller} supplied parent phase from caller-${fixture.caller}.md and resume it on the selected working-tree diff against origin/main. This excerpt comes from ${fixture.runtime}/${fixture.caller}/SKILL.md; resolve installed-relative references there, not from the excerpt file or product directory. That path identifies the asset base, not another entrypoint: do not read or invoke the full parent SKILL.md or rerun its preamble. Earlier preamble/branch/base setup is complete; use the existing local origin/main ref without fetch. Earlier-phase asset locators are ${fixture.runtime}/review/checklist.md and ${fixture.runtime}/qa/templates/functional-report-template.md. Read those files directly when referenced; recursive Glob does not follow the installed asset symlinks. Cross-project learnings are configured off in this owned fixture. ${fixture.reviewStart ? `The actual review-start helper already captured REVIEW_START=${fixture.reviewStart} for this unchanged core pass; retain that token. ` : ''} This fixture evaluates only the supplied parent phase, not later publication stages. The fixture owner recorded these read-only observations when this phase began; they are current until an input changes, so use them instead of re-running those commands, and recheck freshness before completion outputs:\n${fixture.startObservations}\n Read README.md for the project contract and commands. Use diagnostic-client commands such as \`bun scripts/probe.ts <literal>\` for exploratory discoveries and their checkpoint evidence. A required \`bun run test\` is separate suite verification: report it as verification, never as a diagnostic observation or checkpoint anchor/target. Use the production evidence helper to publish each diagnostic checkpoint as \`reports/exploration-NNN.json\`, not inside a nested directory; do not transcribe its observed payload. ${fixture.caseId === 'ship-exploratory-plan-checks' ? 'The previously discovered plan is PLAN.md.' : 'No plan file was found.'} There is no remote service and no release publication is authorized. There is no interactive approver; do not invent answers or permission. Keep normal parent decision gates. Before the completion report and each completed:true review record, read HANDOFF.md and reports/HANDOFF.md if present for any concurrent collaborator update, await the results, and compare evidence with current inputs. A gstack-review-log completed:true record is a completion, not preliminary bookkeeping; a later handoff read cannot validate an earlier completion.\n\nDeadline bookkeeping additionally permits \`bun ${fixture.runtime}/bin/gstack-qa-deadline start ${fixture.cwd}/reports/deadline.json SECONDS [EARLIER_UTC]\`, \`bun ${fixture.runtime}/bin/gstack-qa-deadline status ${fixture.cwd}/reports/deadline.json\`, and \`bun ${fixture.runtime}/bin/gstack-qa-deadline run ${fixture.cwd}/reports/deadline.json -- bun scripts/probe.ts [literal]\`. These are closed literal forms: SECONDS must be positive and at most 300, EARLIER_UTC is the optional caller absolute deadline: use the section clock's Hard deadline UTC, never its Runner entry UTC, reserve-start time or a clock-read time. The child is only the existing diagnostic client with zero or one literal argument. Resolve these exact helper and state paths; do not use variables, another helper, another state file, nested wrappers, scripts, operators or substitutions. Only this helper may create or change reports/deadline.json and its .qa-deadline- temporary files; never use Write/Edit/MultiEdit on those paths. Record the full outer run command in checkpoints and evidence; keep the unchanged child JSON as observed, separate from prefixed guard diagnostics. A completed expired guard-run is not a probe or a pass: retain its unused checkpoint, report not-run coverage and do not restart the deadline. Keep the 12-probe smoke limit. Required suites and explicit plan checks are outside the bounded smoke budget, not permission to reset it.\n\nFunctional evidence uses the same production helper and existing diagnostic client: \`bun ${fixture.runtime}/bin/gstack-qa-evidence capture ${fixture.cwd}/reports NNN --public --deadline ${fixture.cwd}/reports/deadline.json -- bun scripts/probe.ts [literal]\`. These diagnostic receipts are declared public/synthetic, so --public is approved; a fresh three-digit ID is required each time. Explicit plan probes outside the smoke budget may replace --deadline with --timeout-ms 10000; this does not reset or bypass the smoke deadline. Publish causal intent with \`bun ${fixture.runtime}/bin/gstack-qa-evidence checkpoint ${fixture.cwd}/reports NNN CAPTURE_ID 'full prior capture command' 'causal hypothesis' 'full next capture command'\`; quote arguments literally. For complex quoting, Write only capture, observationCommand, hypothesis and nextCommand to reports/intent.json; publish with the same helper: checkpoint REPORT_ROOT NNN intent.json. Materialize is supported when evidence.json Line truncated
appendSystemPrompt: `Caller execution scheduling (fixture contract):
This session has at most 25 assistant turns, including required verification and final artifacts. The command boundary applies to each Bash call, not to the number of independent tool calls in an assistant turn.
After required clock and approval prerequisites settle, issue independent source Reads and read-only discovery together as separate native tool calls once their paths and inputs are known. For example, after the first clock read, one response can Read the caller excerpt, README.md, every product file and the referenced checklist and templates. Wait for their results before decisions that depend on them.