mirror of
https://github.com/garrytan/gstack.git
synced 2026-10-02 17:40:02 +02:00
test: start the eng batching eval with its setup prerequisites supplied
Routing setup and cross-project learnings (D1/D2 in run 36385945043) are never counted and are not what the case measures. The registration now uses the runner's existing preconfiguredReviewActor so the attempt starts at the review; engSetupAUQ still vetoes any late setup question. The registration test pins the option.
This commit is contained in:
1 parent
6f998e4299
commit
2506c4728e
2 files changed
+8
-4
No files matched your search
@@ -451,14 +451,14 @@ test.each([
|
||||
import { describe, expect, mock } from 'bun:test';
|
||||
import * as fs from 'node:fs';
|
||||
import * as real from ${JSON.stringify(runner)};
|
||||
const facts = { runs: 0, stops: [] as boolean[], ceiling: 0 };
|
||||
const facts = { runs: 0, stops: [] as boolean[], ceiling: 0, preconfigured: false };
|
||||
const save = () => fs.writeFileSync(${JSON.stringify(factsPath)}, JSON.stringify(facts));
|
||||
const fp = (signature: string, preReview: boolean, administrative?: string) => ({ signature, preReview, administrative, promptSnippet: signature, options: [], observedAtMs: 1 });
|
||||
mock.module(${JSON.stringify(path.join(ROOT, 'test/helpers/e2e-gate.ts'))}, () => ({
|
||||
describeE2ETier: (tier: string) => { expect(tier).toBe('periodic'); return describe; },
|
||||
}));
|
||||
mock.module(${JSON.stringify(runner)}, () => ({ ...real, runPlanSkillCounting: async (opts: any) => {
|
||||
facts.runs++; facts.ceiling = opts.reviewCountCeiling;
|
||||
facts.runs++; facts.ceiling = opts.reviewCountCeiling; facts.preconfigured = opts.preconfiguredReviewActor;
|
||||
const setup = [fp('s1', true), fp('s2', true)];
|
||||
const review = [fp('r1', false), fp('r2', false), fp('r3', false)];
|
||||
facts.stops = [
|
||||
@@ -479,7 +479,7 @@ await import(${JSON.stringify(path.join(ROOT, 'test/skill-e2e-plan-eng-multi-fin
|
||||
const [exit, out, err] = await Promise.all([child.exited, new Response(child.stdout).text(), new Response(child.stderr).text()]);
|
||||
const facts = JSON.parse(fs.readFileSync(factsPath, 'utf8'));
|
||||
expect(exit, out + err).toBe(passes ? 0 : 1);
|
||||
expect(facts).toEqual({ runs: 1, stops: [false, false, true], ceiling: 7 });
|
||||
expect(facts).toEqual({ runs: 1, stops: [false, false, true], ceiling: 7, preconfigured: true });
|
||||
} finally {
|
||||
fs.rmSync(temp, { recursive: true, force: true });
|
||||
}
|
||||
|
||||
@@ -22,7 +22,7 @@
|
||||
* This is the tightest regression test for the original bug class —
|
||||
* not a band-around-N test, but a "did the agent batch?" test.
|
||||
*
|
||||
* Tier: periodic (~7 min observed; 25 min budget). Sequential by default.
|
||||
* Tier: periodic (~6 min expected; 25 min budget). Sequential by default.
|
||||
*/
|
||||
|
||||
import { test } from 'bun:test';
|
||||
@@ -88,6 +88,10 @@ describeE2E('/plan-eng-review multi-finding batching regression (periodic)', ()
|
||||
isCollectionComplete: (_transcript, fingerprints) =>
|
||||
fingerprints.filter(fp => !fp.preReview && !fp.administrative).length >= FLOOR,
|
||||
reviewCountCeiling: N + 3, // hard cap above floor + tolerance
|
||||
// Supplied prerequisites: routing setup and cross-project learnings are
|
||||
// already declined, so the attempt starts at the review (setup answers
|
||||
// were never counted; engSetupAUQ still vetoes any late setup question).
|
||||
preconfiguredReviewActor: true,
|
||||
timeoutMs: 1_500_000, // 25 min
|
||||
env: { QUESTION_TUNING: 'false', EXPLAIN_LEVEL: 'default' },
|
||||
});
|
||||
|
||||
Reference in new issue
Block a user