mirror of
https://github.com/garrytan/gstack.git
synced 2026-10-03 09:56:57 +02:00
test: clean up the paid eval lane (B1-B4, B6, B7)
- B1: delete paid files that assert nothing or cannot pass meaningfully: skill-llm-eval-spec and skill-e2e-spec-execute (test.todo), gemini-e2e (+ gemini-session-runner; no gemini CLI in CI), ship-idempotency (red since v1.63), the two opus-4-7 *-sonnet overlay wrappers, conductor-prose (+ its source-evaluation replay), codex-e2e-plan-format; drop their keys, scripts and census rows. - B2: skill-llm-eval grades browse/sections/command-list.md with one union judge that also carries the baseline score pin; regression-vs-baseline deleted (paid run: pass, c4/c4/a4). - B3: memory-pipeline, ios-qa, ios-qa-swift-build and plan-tune-cathedral make no model calls; renamed out of the paid glob so they run on every PR. Swift builds need GSTACK_TEST_SWIFT=1; device stub deleted. - B4: codex-e2e*, outside-voice, aside and ios-device cannot run in the CI image; excluded from the weekly lane with a tracked re-entry condition. - B6: fold opus-47's negative routing controls into skill-routing-e2e journey-negatives (paid run: 3/3 unrouted) and delete the file. - B7: delete the never-green brain-privacy-gate eval; a free gstack-skill-start test now proves consent precedes artifacts egress.
This commit is contained in:
1 parent
5d032ef299
commit
53e7f3212f
58 files changed
+273
-2594
No files matched your search
@@ -590,6 +590,52 @@ export default app;
|
||||
}
|
||||
}, CAPTURE_MS);
|
||||
|
||||
testIfSelected('journey-negatives', async () => {
|
||||
// Casual or off-topic prompts that share routing keywords ("wtf",
|
||||
// "algorithm", "send it to the team") must not invoke a skill. Folded
|
||||
// from the retired Opus 4.7 routing eval; same bound: at most one of the
|
||||
// three may route.
|
||||
const cases = [
|
||||
{ name: 'neg-syntax-q', prompt: 'wtf does this Python list comprehension syntax even mean, [x for x in y if z]?' },
|
||||
{ name: 'neg-algo-q', prompt: 'does this bubble sort algorithm actually work in O(n log n)?' },
|
||||
{ name: 'neg-slack-send', prompt: 'can you help me write the slack message? I want to send it to the team.' },
|
||||
];
|
||||
const tmpDir = createRoutingWorkDir('negatives');
|
||||
try {
|
||||
const results = await Promise.all(cases.map(async c => {
|
||||
const result = await runSkillTest({
|
||||
prompt: c.prompt,
|
||||
workingDirectory: tmpDir,
|
||||
maxTurns: 2,
|
||||
allowedTools: ['Skill', 'Read'],
|
||||
timeout: JUDGE_MS,
|
||||
testName: `journey-negatives-${c.name}`,
|
||||
runId,
|
||||
});
|
||||
const skillCalls = result.toolCalls.filter(tc => tc.tool === 'Skill');
|
||||
const actualSkill = skillCalls.length > 0 ? skillCalls[0]?.input?.skill : undefined;
|
||||
logCost(`journey: journey-negatives ${c.name}`, result);
|
||||
evalCollector?.addTest({
|
||||
name: `journey-negatives-${c.name}`,
|
||||
suite: 'Skill Routing E2E',
|
||||
tier: 'e2e',
|
||||
passed: actualSkill === undefined,
|
||||
duration_ms: result.duration,
|
||||
cost_usd: result.costEstimate.estimatedCost,
|
||||
transcript: result.transcript,
|
||||
output: `routed=${actualSkill ?? '(none)'}`,
|
||||
turns_used: result.costEstimate.turnsUsed,
|
||||
exit_reason: result.exitReason,
|
||||
});
|
||||
return { name: c.name, actualSkill };
|
||||
}));
|
||||
const routed = results.filter(r => r.actualSkill !== undefined);
|
||||
expect(routed.length, `negatives routed: ${routed.map(r => `${r.name}→${r.actualSkill}`).join(', ')}`).toBeLessThanOrEqual(1);
|
||||
} finally {
|
||||
fs.rmSync(tmpDir, { recursive: true, force: true });
|
||||
}
|
||||
}, CAPTURE_MS);
|
||||
|
||||
testIfSelected('journey-visual-qa', async () => {
|
||||
const tmpDir = createRoutingWorkDir('visual-qa');
|
||||
try {
|
||||
|
||||
Reference in new issue
Block a user