test: start the eng batching eval with its setup prerequisites supplied

Routing setup and cross-project learnings (D1/D2 in run 36385945043) are
never counted and are not what the case measures. The registration now uses
the runner's existing preconfiguredReviewActor so the attempt starts at the
review; engSetupAUQ still vetoes any late setup question. The registration
test pins the option.
This commit is contained in:
garrytan committed 2026-09-29 15:32:07 +00:00
1 parent 6f998e4299
commit 2506c4728e
2 files changed
+8 -4

No files matched your search

+3 -3
View File
@@ -451,14 +451,14 @@ test.each([
import { describe, expect, mock } from 'bun:test';
import * as fs from 'node:fs';
import * as real from ${JSON.stringify(runner)};
const facts = { runs: 0, stops: [] as boolean[], ceiling: 0 };
const facts = { runs: 0, stops: [] as boolean[], ceiling: 0, preconfigured: false };
const save = () => fs.writeFileSync(${JSON.stringify(factsPath)}, JSON.stringify(facts));
const fp = (signature: string, preReview: boolean, administrative?: string) => ({ signature, preReview, administrative, promptSnippet: signature, options: [], observedAtMs: 1 });
mock.module(${JSON.stringify(path.join(ROOT, 'test/helpers/e2e-gate.ts'))}, () => ({
describeE2ETier: (tier: string) => { expect(tier).toBe('periodic'); return describe; },
}));
mock.module(${JSON.stringify(runner)}, () => ({ ...real, runPlanSkillCounting: async (opts: any) => {
facts.runs++; facts.ceiling = opts.reviewCountCeiling;
facts.runs++; facts.ceiling = opts.reviewCountCeiling; facts.preconfigured = opts.preconfiguredReviewActor;
const setup = [fp('s1', true), fp('s2', true)];
const review = [fp('r1', false), fp('r2', false), fp('r3', false)];
facts.stops = [
@@ -479,7 +479,7 @@ await import(${JSON.stringify(path.join(ROOT, 'test/skill-e2e-plan-eng-multi-fin
const [exit, out, err] = await Promise.all([child.exited, new Response(child.stdout).text(), new Response(child.stderr).text()]);
const facts = JSON.parse(fs.readFileSync(factsPath, 'utf8'));
expect(exit, out + err).toBe(passes ? 0 : 1);
expect(facts).toEqual({ runs: 1, stops: [false, false, true], ceiling: 7 });
expect(facts).toEqual({ runs: 1, stops: [false, false, true], ceiling: 7, preconfigured: true });
} finally {
fs.rmSync(temp, { recursive: true, force: true });
}
@@ -22,7 +22,7 @@
* This is the tightest regression test for the original bug class —
* not a band-around-N test, but a "did the agent batch?" test.
*
* Tier: periodic (~7 min observed; 25 min budget). Sequential by default.
* Tier: periodic (~6 min expected; 25 min budget). Sequential by default.
*/
import { test } from 'bun:test';
@@ -88,6 +88,10 @@ describeE2E('/plan-eng-review multi-finding batching regression (periodic)', ()
isCollectionComplete: (_transcript, fingerprints) =>
fingerprints.filter(fp => !fp.preReview && !fp.administrative).length >= FLOOR,
reviewCountCeiling: N + 3, // hard cap above floor + tolerance
// Supplied prerequisites: routing setup and cross-project learnings are
// already declined, so the attempt starts at the review (setup answers
// were never counted; engSetupAUQ still vetoes any late setup question).
preconfiguredReviewActor: true,
timeoutMs: 1_500_000, // 25 min
env: { QUESTION_TUNING: 'false', EXPLAIN_LEVEL: 'default' },
});