From 2506c4728ee00f072411b73587bda24b46e8e565 Mon Sep 17 00:00:00 2001 From: garrytan Date: Tue, 29 Sep 2026 15:32:07 +0000 Subject: [PATCH] test: start the eng batching eval with its setup prerequisites supplied Routing setup and cross-project learnings (D1/D2 in run 36385945043) are never counted and are not what the case measures. The registration now uses the runner's existing preconfiguredReviewActor so the attempt starts at the review; engSetupAUQ still vetoes any late setup question. The registration test pins the option. --- test/eng-batching-saved-ledger.test.ts | 6 +++--- test/skill-e2e-plan-eng-multi-finding-batching.test.ts | 6 +++++- 2 files changed, 8 insertions(+), 4 deletions(-) diff --git a/test/eng-batching-saved-ledger.test.ts b/test/eng-batching-saved-ledger.test.ts index d9dd7eb57..e7f3d2210 100644 --- a/test/eng-batching-saved-ledger.test.ts +++ b/test/eng-batching-saved-ledger.test.ts @@ -451,14 +451,14 @@ test.each([ import { describe, expect, mock } from 'bun:test'; import * as fs from 'node:fs'; import * as real from ${JSON.stringify(runner)}; -const facts = { runs: 0, stops: [] as boolean[], ceiling: 0 }; +const facts = { runs: 0, stops: [] as boolean[], ceiling: 0, preconfigured: false }; const save = () => fs.writeFileSync(${JSON.stringify(factsPath)}, JSON.stringify(facts)); const fp = (signature: string, preReview: boolean, administrative?: string) => ({ signature, preReview, administrative, promptSnippet: signature, options: [], observedAtMs: 1 }); mock.module(${JSON.stringify(path.join(ROOT, 'test/helpers/e2e-gate.ts'))}, () => ({ describeE2ETier: (tier: string) => { expect(tier).toBe('periodic'); return describe; }, })); mock.module(${JSON.stringify(runner)}, () => ({ ...real, runPlanSkillCounting: async (opts: any) => { - facts.runs++; facts.ceiling = opts.reviewCountCeiling; + facts.runs++; facts.ceiling = opts.reviewCountCeiling; facts.preconfigured = opts.preconfiguredReviewActor; const setup = [fp('s1', true), fp('s2', true)]; const review = [fp('r1', false), fp('r2', false), fp('r3', false)]; facts.stops = [ @@ -479,7 +479,7 @@ await import(${JSON.stringify(path.join(ROOT, 'test/skill-e2e-plan-eng-multi-fin const [exit, out, err] = await Promise.all([child.exited, new Response(child.stdout).text(), new Response(child.stderr).text()]); const facts = JSON.parse(fs.readFileSync(factsPath, 'utf8')); expect(exit, out + err).toBe(passes ? 0 : 1); - expect(facts).toEqual({ runs: 1, stops: [false, false, true], ceiling: 7 }); + expect(facts).toEqual({ runs: 1, stops: [false, false, true], ceiling: 7, preconfigured: true }); } finally { fs.rmSync(temp, { recursive: true, force: true }); } diff --git a/test/skill-e2e-plan-eng-multi-finding-batching.test.ts b/test/skill-e2e-plan-eng-multi-finding-batching.test.ts index 424b3d6f1..7c94803f1 100644 --- a/test/skill-e2e-plan-eng-multi-finding-batching.test.ts +++ b/test/skill-e2e-plan-eng-multi-finding-batching.test.ts @@ -22,7 +22,7 @@ * This is the tightest regression test for the original bug class — * not a band-around-N test, but a "did the agent batch?" test. * - * Tier: periodic (~7 min observed; 25 min budget). Sequential by default. + * Tier: periodic (~6 min expected; 25 min budget). Sequential by default. */ import { test } from 'bun:test'; @@ -88,6 +88,10 @@ describeE2E('/plan-eng-review multi-finding batching regression (periodic)', () isCollectionComplete: (_transcript, fingerprints) => fingerprints.filter(fp => !fp.preReview && !fp.administrative).length >= FLOOR, reviewCountCeiling: N + 3, // hard cap above floor + tolerance + // Supplied prerequisites: routing setup and cross-project learnings are + // already declined, so the attempt starts at the review (setup answers + // were never counted; engSetupAUQ still vetoes any late setup question). + preconfiguredReviewActor: true, timeoutMs: 1_500_000, // 25 min env: { QUESTION_TUNING: 'false', EXPLAIN_LEVEL: 'default' }, });