test: clean up the paid eval lane (B1-B4, B6, B7)

- B1: delete paid files that assert nothing or cannot pass meaningfully:
  skill-llm-eval-spec and skill-e2e-spec-execute (test.todo), gemini-e2e
  (+ gemini-session-runner; no gemini CLI in CI), ship-idempotency (red
  since v1.63), the two opus-4-7 *-sonnet overlay wrappers, conductor-prose
  (+ its source-evaluation replay), codex-e2e-plan-format; drop their keys,
  scripts and census rows.
- B2: skill-llm-eval grades browse/sections/command-list.md with one union
  judge that also carries the baseline score pin; regression-vs-baseline
  deleted (paid run: pass, c4/c4/a4).
- B3: memory-pipeline, ios-qa, ios-qa-swift-build and plan-tune-cathedral
  make no model calls; renamed out of the paid glob so they run on every
  PR. Swift builds need GSTACK_TEST_SWIFT=1; device stub deleted.
- B4: codex-e2e*, outside-voice, aside and ios-device cannot run in the CI
  image; excluded from the weekly lane with a tracked re-entry condition.
- B6: fold opus-47's negative routing controls into skill-routing-e2e
  journey-negatives (paid run: 3/3 unrouted) and delete the file.
- B7: delete the never-green brain-privacy-gate eval; a free
  gstack-skill-start test now proves consent precedes artifacts egress.
This commit is contained in:
garrytan committed 2026-09-29 06:08:49 +00:00
1 parent 5d032ef299
commit 53e7f3212f
58 files changed
+273 -2594

No files matched your search

+46
View File
@@ -590,6 +590,52 @@ export default app;
}
}, CAPTURE_MS);
testIfSelected('journey-negatives', async () => {
// Casual or off-topic prompts that share routing keywords ("wtf",
// "algorithm", "send it to the team") must not invoke a skill. Folded
// from the retired Opus 4.7 routing eval; same bound: at most one of the
// three may route.
const cases = [
{ name: 'neg-syntax-q', prompt: 'wtf does this Python list comprehension syntax even mean, [x for x in y if z]?' },
{ name: 'neg-algo-q', prompt: 'does this bubble sort algorithm actually work in O(n log n)?' },
{ name: 'neg-slack-send', prompt: 'can you help me write the slack message? I want to send it to the team.' },
];
const tmpDir = createRoutingWorkDir('negatives');
try {
const results = await Promise.all(cases.map(async c => {
const result = await runSkillTest({
prompt: c.prompt,
workingDirectory: tmpDir,
maxTurns: 2,
allowedTools: ['Skill', 'Read'],
timeout: JUDGE_MS,
testName: `journey-negatives-${c.name}`,
runId,
});
const skillCalls = result.toolCalls.filter(tc => tc.tool === 'Skill');
const actualSkill = skillCalls.length > 0 ? skillCalls[0]?.input?.skill : undefined;
logCost(`journey: journey-negatives ${c.name}`, result);
evalCollector?.addTest({
name: `journey-negatives-${c.name}`,
suite: 'Skill Routing E2E',
tier: 'e2e',
passed: actualSkill === undefined,
duration_ms: result.duration,
cost_usd: result.costEstimate.estimatedCost,
transcript: result.transcript,
output: `routed=${actualSkill ?? '(none)'}`,
turns_used: result.costEstimate.turnsUsed,
exit_reason: result.exitReason,
});
return { name: c.name, actualSkill };
}));
const routed = results.filter(r => r.actualSkill !== undefined);
expect(routed.length, `negatives routed: ${routed.map(r => `${r.name}→${r.actualSkill}`).join(', ')}`).toBeLessThanOrEqual(1);
} finally {
fs.rmSync(tmpDir, { recursive: true, force: true });
}
}, CAPTURE_MS);
testIfSelected('journey-visual-qa', async () => {
const tmpDir = createRoutingWorkDir('visual-qa');
try {