mirror of
https://github.com/garrytan/gstack.git
synced 2026-10-02 17:40:02 +02:00
The case's only verdict is reviewCount >= FLOOR (3). Run 36385945043 had three distinct acknowledged review decisions at 6m41s but kept answering until the ceiling (7) at 12m13s. The registration now passes the runner's existing isCollectionComplete stop once FLOOR non-setup, non-administrative review decisions are acknowledged; the floor check, ceiling, budget and counter are unchanged. A child-process registration test proves the stop predicate and that below-floor and timeout outcomes still fail.
127 lines
5.7 KiB
TypeScript
127 lines
5.7 KiB
TypeScript
/**
|
|
* /plan-eng-review multi-finding batching regression (periodic, paid, real-PTY).
|
|
*
|
|
* Catches the specific shape of the May 2026 transcript bug that the
|
|
* single-finding gate-tier floor test cannot detect: a model that fires
|
|
* one AskUserQuestion and then batches the remaining findings into a
|
|
* single "## Decisions to confirm" plan write + ExitPlanMode.
|
|
*
|
|
* Why a separate test from skill-e2e-plan-eng-finding-floor:
|
|
* - The gate-tier floor (runPlanSkillFloorCheck) exits on the first AUQ
|
|
* render and returns success. A model that fires once-then-batches
|
|
* would pass that test trivially.
|
|
* - This test uses runPlanSkillCounting at periodic tier (~25 min budget,
|
|
* N-AUQ tracking, ceiling-bounded retries) to actually count distinct
|
|
* review-phase AUQs and assert the model fires one per finding. Collection
|
|
* stops as soon as the floor is proven (~7 min observed).
|
|
*
|
|
* Why a separate test from skill-e2e-plan-eng-finding-count (the existing
|
|
* 5-finding count test):
|
|
* - The fixture here mirrors the D1-D4 transcript shape (4 findings) and
|
|
* the floor matches that exact threshold (3, the [N-1] tolerance band).
|
|
* This is the tightest regression test for the original bug class —
|
|
* not a band-around-N test, but a "did the agent batch?" test.
|
|
*
|
|
* Tier: periodic (~7 min observed; 25 min budget). Sequential by default.
|
|
*/
|
|
|
|
import { test } from 'bun:test';
|
|
import { describeE2ETier } from './helpers/e2e-gate';
|
|
import * as fs from 'node:fs';
|
|
import * as os from 'node:os';
|
|
import * as path from 'node:path';
|
|
import {
|
|
runPlanSkillCounting,
|
|
engStep0Boundary,
|
|
engSetupAUQ,
|
|
engFirstReviewAUQ,
|
|
} from './helpers/claude-pty-runner';
|
|
import { FORCING_BATCHING_ENG } from './fixtures/forcing-finding-seeds';
|
|
import { createEngBatchingIssueCounter } from './helpers/eng-seeded-coverage';
|
|
|
|
const describeE2E = describeE2ETier('periodic');
|
|
|
|
const N = 4;
|
|
const FLOOR = N - 1; // 3 — agent must fire at least one AUQ per non-batched finding
|
|
|
|
/** Plan-file target baked into the FORCING_BATCHING_ENG fixture prompt.
|
|
* Rewritten per-run to a mkdtemp path so concurrent runs (--retry,
|
|
* EVALS_JOBS>1, sibling worktrees) never share one /tmp artifact. */
|
|
const FIXTURE_PLAN_PATH = '/tmp/gstack-test-plan-eng-batching.md';
|
|
|
|
describeE2E('/plan-eng-review multi-finding batching regression (periodic)', () => {
|
|
test(
|
|
`4-finding plan emits >= ${FLOOR} review-phase AskUserQuestions (no batching)`,
|
|
async () => {
|
|
const tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'gstack-e2e-plan-eng-batching-'));
|
|
const planPath = path.join(tmpDir, 'gstack-test-plan-eng-batching.md');
|
|
const followUpPrompt = FORCING_BATCHING_ENG.replaceAll(FIXTURE_PLAN_PATH, planPath);
|
|
if (!followUpPrompt.includes(planPath)) {
|
|
throw new Error(
|
|
`fixture drift: FORCING_BATCHING_ENG no longer contains ${FIXTURE_PLAN_PATH} — update FIXTURE_PLAN_PATH`,
|
|
);
|
|
}
|
|
|
|
try {
|
|
const findings = createEngBatchingIssueCounter(() => {
|
|
try {
|
|
const stat = fs.lstatSync(planPath);
|
|
return stat.isFile() && !stat.isSymbolicLink() ? fs.readFileSync(planPath, 'utf8') : '';
|
|
} catch (error) {
|
|
if ((error as NodeJS.ErrnoException).code === 'ENOENT') return '';
|
|
throw error;
|
|
}
|
|
}, engSetupAUQ);
|
|
const obs = await runPlanSkillCounting({
|
|
skillName: 'plan-eng-review',
|
|
slashCommand: '/plan-eng-review',
|
|
followUpPrompt,
|
|
permissionPlanPath: planPath,
|
|
isLastStep0AUQ: engStep0Boundary,
|
|
isSetupAUQ: engSetupAUQ,
|
|
isFirstReviewAUQ: engFirstReviewAUQ,
|
|
isReviewAUQ: findings.isReviewAUQ,
|
|
// The only verdict is the floor. Once FLOOR distinct acknowledged
|
|
// review decisions exist, a batching regression can no longer occur
|
|
// in this attempt; stop instead of letting the review run to the
|
|
// ceiling (run 36385945043: floor at 6m41s, ceiling at 12m13s).
|
|
isCollectionComplete: (_transcript, fingerprints) =>
|
|
fingerprints.filter(fp => !fp.preReview && !fp.administrative).length >= FLOOR,
|
|
reviewCountCeiling: N + 3, // hard cap above floor + tolerance
|
|
timeoutMs: 1_500_000, // 25 min
|
|
env: { QUESTION_TUNING: 'false', EXPLAIN_LEVEL: 'default' },
|
|
});
|
|
|
|
if (!['plan_ready', 'completion_summary', 'collection_complete', 'ceiling_reached'].includes(obs.outcome)) {
|
|
throw new Error(
|
|
`multi-finding batching test FAILED: outcome=${obs.outcome}\n` +
|
|
`step0=${obs.step0Count} review=${obs.reviewCount} elapsed=${obs.elapsedMs}ms\n` +
|
|
`--- evidence (last 3KB) ---\n${obs.evidence}`,
|
|
);
|
|
}
|
|
if (obs.reviewCount < FLOOR) {
|
|
throw new Error(
|
|
`BATCHING REGRESSION: reviewCount=${obs.reviewCount} < FLOOR=${FLOOR}.\n` +
|
|
`Agent surfaced fewer review-phase AUQs than findings — this is the\n` +
|
|
`May 2026 transcript bug shape: model batched multiple findings into\n` +
|
|
`a single plan write + ExitPlanMode instead of asking one per finding.\n` +
|
|
`Review-phase fingerprints:\n` +
|
|
obs.fingerprints
|
|
.filter((f) => !f.preReview)
|
|
.map((f) => ` - "${f.promptSnippet.slice(0, 80)}"`)
|
|
.join('\n') +
|
|
`\n--- evidence (last 3KB) ---\n${obs.evidence}`,
|
|
);
|
|
}
|
|
} finally {
|
|
try {
|
|
fs.rmSync(tmpDir, { recursive: true, force: true });
|
|
} catch {
|
|
/* best-effort */
|
|
}
|
|
}
|
|
},
|
|
1_500_000 /* physical ceiling: the 25-min CI job + 1800s shard wall cap what can actually execute */,
|
|
);
|
|
});
|