test: run seven paid evals on the current default capture model (B8)

skill-e2e-{auq-matrix,plan-format,qa-bugs,retro,workflow} pinned
claude-opus-4-7 and skill-e2e-office-hours plus -brain-writeback pinned
claude-sonnet-4-6; none tests a historical model, so they now capture with
resolveEvalModel('capture'), and the free harness tests that execute these
registrations receive the same resolver. The paid re-pin run passed all of
them. skill-e2e-{design,office-hours-phase4,plan-prosons,plan} keep
claude-opus-4-7: six of their cases failed on the default model (three
timeouts, a missing report file, a format miss and a posture score of 3), so
per the plan's fallback they keep their pins with a TODOS entry. The pre-spend
estimate and drop threshold are in docs/test-audit-2026-09.md.
This commit is contained in:
garrytan committed 2026-09-29 09:16:19 +00:00
1 parent 799f6ec36c
commit 67084a5caf
11 files changed
+55 -23

No files matched your search

+9 -8
View File
@@ -18,6 +18,7 @@
* accordingly.
*/
import { describe, test, expect, beforeAll, afterAll } from 'bun:test';
import { resolveEvalModel } from '../lib/eval-model';
import { CAPTURE_MS } from './helpers/eval-budgets';
import { runSkillTest } from './helpers/session-runner';
import { runRecordedOfficeHoursAttempt, OFFICE_HOURS_BUN_GRACE_MS } from './helpers/office-hours-attempt';
@@ -128,7 +129,7 @@ describeIfSelected('Plan Format — CEO Mode Selection', ['plan-ceo-review-forma
const judgeMetadata: Pick<EvalTestEntry, 'judge_scores' | 'judge_reasoning'> = {};
await runRecordedOfficeHoursAttempt({
collector: evalCollector, name: '/plan-ceo-review-format-mode', suite: 'Plan Format — CEO Mode Selection',
model: 'claude-opus-4-7', budgetMs: CAPTURE_MS,
model: resolveEvalModel('capture'), budgetMs: CAPTURE_MS,
judgeMetadata,
run: signal => runSkillTest({
signal,
@@ -146,7 +147,7 @@ After writing the file, stop. Do not continue the review.`,
timeout: CAPTURE_MS,
testName: 'plan-ceo-review-format-mode',
runId,
model: 'claude-opus-4-7',
model: resolveEvalModel('capture'),
}),
validate: async (result, signal) => {
logCost('/plan-ceo-review format (mode)', result);
@@ -195,7 +196,7 @@ describeIfSelected('Plan Format — CEO Approach Menu', ['plan-ceo-review-format
const judgeMetadata: Pick<EvalTestEntry, 'judge_scores' | 'judge_reasoning'> = {};
await runRecordedOfficeHoursAttempt({
collector: evalCollector, name: '/plan-ceo-review-format-approach', suite: 'Plan Format — CEO Approach Menu',
model: 'claude-opus-4-7', budgetMs: CAPTURE_MS,
model: resolveEvalModel('capture'), budgetMs: CAPTURE_MS,
judgeMetadata,
run: signal => runSkillTest({
signal,
@@ -213,7 +214,7 @@ After writing the file, stop. Do not continue the review.`,
timeout: CAPTURE_MS,
testName: 'plan-ceo-review-format-approach',
runId,
model: 'claude-opus-4-7',
model: resolveEvalModel('capture'),
}),
validate: async (result, signal) => {
logCost('/plan-ceo-review format (approach)', result);
@@ -261,7 +262,7 @@ describeIfSelected('Plan Format — Eng Coverage Issue', ['plan-eng-review-forma
const judgeMetadata: Pick<EvalTestEntry, 'judge_scores' | 'judge_reasoning'> = {};
await runRecordedOfficeHoursAttempt({
collector: evalCollector, name: '/plan-eng-review-format-coverage', suite: 'Plan Format — Eng Coverage Issue',
model: 'claude-opus-4-7', budgetMs: CAPTURE_MS,
model: resolveEvalModel('capture'), budgetMs: CAPTURE_MS,
judgeMetadata,
run: signal => runSkillTest({
signal,
@@ -282,7 +283,7 @@ After writing the file with that ONE question, stop. Do not continue the review.
timeout: CAPTURE_MS,
testName: 'plan-eng-review-format-coverage',
runId,
model: 'claude-opus-4-7',
model: resolveEvalModel('capture'),
}),
validate: async (result, signal) => {
logCost('/plan-eng-review format (coverage)', result);
@@ -330,7 +331,7 @@ describeIfSelected('Plan Format — Eng Kind Issue', ['plan-eng-review-format-ki
const judgeMetadata: Pick<EvalTestEntry, 'judge_scores' | 'judge_reasoning'> = {};
await runRecordedOfficeHoursAttempt({
collector: evalCollector, name: '/plan-eng-review-format-kind', suite: 'Plan Format — Eng Kind Issue',
model: 'claude-opus-4-7', budgetMs: CAPTURE_MS,
model: resolveEvalModel('capture'), budgetMs: CAPTURE_MS,
judgeMetadata,
run: signal => runSkillTest({
signal,
@@ -348,7 +349,7 @@ After writing the file with that ONE question, stop. Do not continue the review.
timeout: CAPTURE_MS,
testName: 'plan-eng-review-format-kind',
runId,
model: 'claude-opus-4-7',
model: resolveEvalModel('capture'),
}),
validate: async (result, signal) => {
logCost('/plan-eng-review format (kind)', result);