mirror of
https://github.com/garrytan/gstack.git
synced 2026-10-02 17:40:02 +02:00
A completed GSTACK REVIEW REPORT ends the review, so the review-question count is final there. Run 36606688266 wrote its report at 1,248 s and closed the session at 1,318 s; the case now stops collection and applies the unchanged floor at the report instead of waiting out the session. No budget changes.
137 lines
6.3 KiB
TypeScript
137 lines
6.3 KiB
TypeScript
/**
|
|
* /plan-eng-review multi-finding batching regression (periodic, paid, real-PTY).
|
|
*
|
|
* Catches the specific shape of the May 2026 transcript bug that the
|
|
* single-finding gate-tier floor test cannot detect: a model that fires
|
|
* one AskUserQuestion and then batches the remaining findings into a
|
|
* single "## Decisions to confirm" plan write + ExitPlanMode.
|
|
*
|
|
* Why a separate test from skill-e2e-plan-eng-finding-floor:
|
|
* - The gate-tier floor (runPlanSkillFloorCheck) exits on the first AUQ
|
|
* render and returns success. A model that fires once-then-batches
|
|
* would pass that test trivially.
|
|
* - This test uses runPlanSkillCounting at periodic tier (~25 min budget,
|
|
* N-AUQ tracking, ceiling-bounded retries) to actually count distinct
|
|
* review-phase AUQs and assert the model fires one per finding. Collection
|
|
* stops as soon as the floor is proven (~7 min observed).
|
|
*
|
|
* Why a separate test from skill-e2e-plan-eng-finding-count (the existing
|
|
* 5-finding count test):
|
|
* - The fixture here mirrors the D1-D4 transcript shape (4 findings) and
|
|
* the floor matches that exact threshold (3, the [N-1] tolerance band).
|
|
* This is the tightest regression test for the original bug class —
|
|
* not a band-around-N test, but a "did the agent batch?" test.
|
|
*
|
|
* Tier: periodic (~6 min expected; 25 min budget). Sequential by default.
|
|
*/
|
|
|
|
import { test } from 'bun:test';
|
|
import { describeE2ETier } from './helpers/e2e-gate';
|
|
import * as fs from 'node:fs';
|
|
import * as os from 'node:os';
|
|
import * as path from 'node:path';
|
|
import {
|
|
runPlanSkillCounting,
|
|
engStep0Boundary,
|
|
engSetupAUQ,
|
|
engFirstReviewAUQ,
|
|
hasCompletePlanReport,
|
|
} from './helpers/claude-pty-runner';
|
|
import { FORCING_BATCHING_ENG } from './fixtures/forcing-finding-seeds';
|
|
import { createEngBatchingIssueCounter } from './helpers/eng-seeded-coverage';
|
|
|
|
const describeE2E = describeE2ETier('periodic');
|
|
|
|
const N = 4;
|
|
const FLOOR = N - 1; // 3 — agent must fire at least one AUQ per non-batched finding
|
|
|
|
/** Plan-file target baked into the FORCING_BATCHING_ENG fixture prompt.
|
|
* Rewritten per-run to a mkdtemp path so concurrent runs (--retry,
|
|
* EVALS_JOBS>1, sibling worktrees) never share one /tmp artifact. */
|
|
const FIXTURE_PLAN_PATH = '/tmp/gstack-test-plan-eng-batching.md';
|
|
|
|
describeE2E('/plan-eng-review multi-finding batching regression (periodic)', () => {
|
|
test(
|
|
`4-finding plan emits >= ${FLOOR} review-phase AskUserQuestions (no batching)`,
|
|
async () => {
|
|
const startedAt = Date.now();
|
|
const tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'gstack-e2e-plan-eng-batching-'));
|
|
const planPath = path.join(tmpDir, 'gstack-test-plan-eng-batching.md');
|
|
const followUpPrompt = FORCING_BATCHING_ENG.replaceAll(FIXTURE_PLAN_PATH, planPath);
|
|
if (!followUpPrompt.includes(planPath)) {
|
|
throw new Error(
|
|
`fixture drift: FORCING_BATCHING_ENG no longer contains ${FIXTURE_PLAN_PATH} — update FIXTURE_PLAN_PATH`,
|
|
);
|
|
}
|
|
|
|
try {
|
|
const findings = createEngBatchingIssueCounter(() => {
|
|
try {
|
|
const stat = fs.lstatSync(planPath);
|
|
return stat.isFile() && !stat.isSymbolicLink() ? fs.readFileSync(planPath, 'utf8') : '';
|
|
} catch (error) {
|
|
if ((error as NodeJS.ErrnoException).code === 'ENOENT') return '';
|
|
throw error;
|
|
}
|
|
}, engSetupAUQ);
|
|
const obs = await runPlanSkillCounting({
|
|
skillName: 'plan-eng-review',
|
|
slashCommand: '/plan-eng-review',
|
|
followUpPrompt,
|
|
permissionPlanPath: planPath,
|
|
isLastStep0AUQ: engStep0Boundary,
|
|
isSetupAUQ: engSetupAUQ,
|
|
isFirstReviewAUQ: engFirstReviewAUQ,
|
|
isReviewAUQ: findings.isReviewAUQ,
|
|
// The only verdict is the floor. Once FLOOR distinct acknowledged
|
|
// review decisions exist, a batching regression can no longer occur
|
|
// in this attempt; stop instead of letting the review run to the
|
|
// ceiling (run 36385945043: floor at 6m41s, ceiling at 12m13s).
|
|
// A completed report ends the review, so the count is final there:
|
|
// grade it now rather than waiting out the session (run 36606688266
|
|
// wrote its report at 1,248 s and closed at 1,318 s).
|
|
isCollectionComplete: (_transcript, fingerprints) =>
|
|
fingerprints.filter(fp => !fp.preReview && !fp.administrative).length >= FLOOR ||
|
|
hasCompletePlanReport(planPath, startedAt, Date.now()),
|
|
reviewCountCeiling: N + 3, // hard cap above floor + tolerance
|
|
// Supplied prerequisites: routing setup and cross-project learnings are
|
|
// already declined, so the attempt starts at the review (setup answers
|
|
// were never counted; engSetupAUQ still vetoes any late setup question).
|
|
preconfiguredReviewActor: true,
|
|
timeoutMs: 1_500_000, // 25 min
|
|
env: { QUESTION_TUNING: 'false', EXPLAIN_LEVEL: 'default' },
|
|
});
|
|
|
|
if (!['plan_ready', 'completion_summary', 'collection_complete', 'ceiling_reached'].includes(obs.outcome)) {
|
|
throw new Error(
|
|
`multi-finding batching test FAILED: outcome=${obs.outcome}\n` +
|
|
`step0=${obs.step0Count} review=${obs.reviewCount} elapsed=${obs.elapsedMs}ms\n` +
|
|
`--- evidence (last 3KB) ---\n${obs.evidence}`,
|
|
);
|
|
}
|
|
if (obs.reviewCount < FLOOR) {
|
|
throw new Error(
|
|
`BATCHING REGRESSION: reviewCount=${obs.reviewCount} < FLOOR=${FLOOR}.\n` +
|
|
`Agent surfaced fewer review-phase AUQs than findings — this is the\n` +
|
|
`May 2026 transcript bug shape: model batched multiple findings into\n` +
|
|
`a single plan write + ExitPlanMode instead of asking one per finding.\n` +
|
|
`Review-phase fingerprints:\n` +
|
|
obs.fingerprints
|
|
.filter((f) => !f.preReview)
|
|
.map((f) => ` - "${f.promptSnippet.slice(0, 80)}"`)
|
|
.join('\n') +
|
|
`\n--- evidence (last 3KB) ---\n${obs.evidence}`,
|
|
);
|
|
}
|
|
} finally {
|
|
try {
|
|
fs.rmSync(tmpDir, { recursive: true, force: true });
|
|
} catch {
|
|
/* best-effort */
|
|
}
|
|
}
|
|
},
|
|
1_500_000 /* physical ceiling: the 25-min CI job + 1800s shard wall cap what can actually execute */,
|
|
);
|
|
});
|