From 4c5fc0e2ce1bafe0f8492c8463b13c4255d9c351 Mon Sep 17 00:00:00 2001 From: garrytan Date: Tue, 29 Sep 2026 17:28:32 +0000 Subject: [PATCH] test(qa-bugs): keep claude-opus-4-7 after qa-b6-static stalled on the default model qa-b6-static timed out on claude-fable-5-1 in census 36597762183 and in one of two targeted reruns. Both times the stream stopped mid-message with no pending tool, right after the model found the disabled submit button, and stayed silent until the 300 s deadline. Per the B8 fallback, re-pin with a TODOS entry; budgets and retries are unchanged. A rerun on opus-4-7 passed (125 s, 5/5 detected). --- TODOS.md | 7 +++++-- test/skill-e2e-qa-bugs.test.ts | 3 +-- 2 files changed, 6 insertions(+), 4 deletions(-) diff --git a/TODOS.md b/TODOS.md index 79737cd06..ce8fae2ee 100644 --- a/TODOS.md +++ b/TODOS.md @@ -843,13 +843,16 @@ and `test/dx-selected-navigation-ap.test.ts`. One shared table run once against only after `engFirstReviewAUQ` checks native completion once at entry; today each branch gates it separately, so the change alters a paid verdict and needs its own paid run. -### P3: Re-pin the four remaining claude-opus-4-7 paid files +### P3: Re-pin the five remaining claude-opus-4-7 paid files **What:** The 2026-09 audit moved seven paid evals to the default capture model (`resolveEvalModel('capture')`). `skill-e2e-design`, `skill-e2e-office-hours-phase4`, `skill-e2e-plan-prosons` and `skill-e2e-plan` keep `claude-opus-4-7` because six cases failed on the default model in one run (plan-design-review-plan-mode timeout, office-hours-phase4-fork format, plan-review-prosons-neutral-neg missing output, plan-ceo-review-selective and -plan-eng-review 600 s timeouts, plan-ceo-review-expansion-energy posture score 3). They measure an old model. +plan-eng-review 600 s timeouts, plan-ceo-review-expansion-energy posture score 3). `skill-e2e-qa-bugs` returned +to `claude-opus-4-7` after `qa-b6-static` timed out on the default model in two of three runs (census 36597762183 +and a targeted local rerun): each time the stream stopped mid-message, with no pending tool, right after the model +found the disabled submit button, and emitted nothing until the 300 s case deadline. They measure an old model. **Re-entry:** fix the prompt, budget or rubric so each case passes on the default model in one run, then drop the pin. diff --git a/test/skill-e2e-qa-bugs.test.ts b/test/skill-e2e-qa-bugs.test.ts index 321477f02..14db97bf6 100644 --- a/test/skill-e2e-qa-bugs.test.ts +++ b/test/skill-e2e-qa-bugs.test.ts @@ -1,5 +1,4 @@ import { describe, test, expect, beforeAll, afterAll } from 'bun:test'; -import { resolveEvalModel } from '../lib/eval-model'; import { CAPTURE_MS, CAPTURE_LONG_MS } from './helpers/eval-budgets'; import { runSkillTest } from './helpers/session-runner'; import { outcomeJudge } from './helpers/llm-judge'; @@ -109,7 +108,7 @@ CRITICAL RULES: timeout: CAPTURE_MS, testName: `qa-${label}`, runId, - model: resolveEvalModel('capture'), + model: 'claude-opus-4-7', }); logCost(`/qa ${label}`, result);