diff --git a/docs/test-audit-2026-09.md b/docs/test-audit-2026-09.md new file mode 100644 index 000000000..cbac82ef6 --- /dev/null +++ b/docs/test-audit-2026-09.md @@ -0,0 +1,23 @@ +# Test audit 2026-09: evidence + +Evidence for the test-reduction branch (plan approved via /autoplan). Sections are added by the commit they support. + +## B8 pre-spend estimate (recorded 2026-09-29, before any B8 paid run) + +Source: latest weekly periodic artifacts (runs 36385945043 = 09-28, 35567915613 = 09-21), per-shard eval JSON cost_usd. +Price ratio from test/helpers/pricing.ts: claude-fable-5-1 (default capture, lib/eval-model.ts) $10/$50 per MTok in/out; +claude-opus-4-7 $15/$75 (ratio 0.667 on both); claude-sonnet-4-6 $3/$15 (ratio 3.33 on both). + +| Files | Old pin | Weekly $ (09-28) | Est. weekly $ on default | Delta | +|---|---|---:|---:|---:| +| plan, design, plan-prosons, plan-format, qa-bugs, retro, office-hours-phase4 | opus-4-7 | 15.78 | 10.52 | −5.26 | +| office-hours, office-hours-brain-writeback | sonnet-4-6 | 0.91 | 3.03 | +2.12 | +| auq-matrix, workflow | opus-4-7 | no result in the retained artifacts | — | ≤ 0 (ratio 0.667) | +| **B8 total** | | 16.69 | 13.55 | **−3.14** | + +Assumes the same token volume per case (a verbosity change moves this; the ratio applies to input and output alike). +Wall clock: unchanged shard walls (budgets do not depend on model). Drop threshold, fixed now: B8 is dropped from this PR +if its estimated net weekly dollars after C and B5 savings are above zero. Estimated net: −3.14 (B8) − C savings +(five retired evals) − B5 savings (18 hollow shards, 23 census judges) < 0 → B8 proceeds to its one paid run. +Fallback check: `git log -S claude-sonnet-4-6` on skill-e2e-office-hours and -brain-writeback shows only 636175d / #2264 +(infra hardening), no cost rationale → both re-pinned. diff --git a/test/office-hours-attempt.test.ts b/test/office-hours-attempt.test.ts index 6ddc3e9ff..94a25295f 100644 --- a/test/office-hours-attempt.test.ts +++ b/test/office-hours-attempt.test.ts @@ -1,5 +1,6 @@ /** Free recording fixtures; every runner and judge below is synthetic. */ import { describe, expect, spyOn, test } from 'bun:test'; +import { resolveEvalModel } from '../lib/eval-model'; import * as fs from 'node:fs'; import * as os from 'node:os'; import * as path from 'node:path'; @@ -649,7 +650,8 @@ describe('Plan format actual capture and judge lifecycle', () => { expect(output).not.toContain('Unhandled error between tests'); const starts = events.filter(event => event.kind === 'start'); expect(starts.map(({ timeout, maxTurns, model }) => ({ timeout, maxTurns, model }))) - .toEqual([1, 2].map(() => ({ timeout: 300, maxTurns: 10, model: 'claude-opus-4-7' }))); + .toEqual([1, 2].map(() => ({ timeout: 300, maxTurns: 10, + model: file.includes('plan-prosons') ? 'claude-opus-4-7' : resolveEvalModel('capture') }))); expect(events.filter(event => event.kind === 'ready').map(event => event.fixtureExists)).toEqual([true, true]); expect(entries).toHaveLength(2); expect(entries.map(entry => entry.attempt)).toEqual([1, 2]); diff --git a/test/office-hours-writeback-env.test.ts b/test/office-hours-writeback-env.test.ts index f1b6d8ef2..6c01aca10 100644 --- a/test/office-hours-writeback-env.test.ts +++ b/test/office-hours-writeback-env.test.ts @@ -61,7 +61,7 @@ console.log(JSON.stringify({ type: 'result', subtype: 'success', is_error: false // payload assertions. The negative case restores only the original typo. expect(suiteSource).toContain('env: childEnv,'); let selectedSource = option === 'env' ? suiteSource : suiteSource.replace('env: childEnv,', 'extraEnv: childEnv,'); - selectedSource = selectedSource.replace(/from '(\.\/helpers\/[^']+)'/g, + selectedSource = selectedSource.replace(/from '((?:\.\/helpers|\.\.\/lib)\/[^']+)'/g, (_match, spec: string) => `from ${JSON.stringify(path.resolve(ROOT, 'test', spec))}`); const suiteCopy = path.join(dir, 'writeback-suite.ts'); fs.writeFileSync(suiteCopy, selectedSource); diff --git a/test/office-posture-recording.test.ts b/test/office-posture-recording.test.ts index 888387aa7..8527adece 100644 --- a/test/office-posture-recording.test.ts +++ b/test/office-posture-recording.test.ts @@ -1,4 +1,5 @@ import { expect, test } from 'bun:test'; +import { resolveEvalModel } from '../lib/eval-model'; import * as fs from 'node:fs'; import * as path from 'node:path'; import { CAPTURE_MS, CAPTURE_LONG_MS } from './helpers/eval-budgets'; @@ -26,7 +27,7 @@ async function exercise(owner: Owner, scenarios: Scenario[]) { } }; const args: Record = { expect, beforeAll: (fn: () => void) => setups.push(fn), afterAll: (fn: () => void) => finalizers.push(fn), - CAPTURE_MS, CAPTURE_LONG_MS, OFFICE_HOURS_BUN_GRACE_MS, runRecordedOfficeHoursAttempt, + CAPTURE_MS, CAPTURE_LONG_MS, OFFICE_HOURS_BUN_GRACE_MS, runRecordedOfficeHoursAttempt, resolveEvalModel, ROOT: '/source', runId: 'synthetic-posture-run', evalsEnabled: true, describeIfSelected: (_title: string, _names: string[], fn: () => void) => fn(), testConcurrentIfSelected: (name: string, fn: () => Promise, timeout: number) => { expect(timeout).toBe(CAPTURE_LONG_MS + OFFICE_HOURS_BUN_GRACE_MS); callbacks.set(name, fn); }, @@ -44,7 +45,7 @@ async function exercise(owner: Owner, scenarios: Scenario[]) { runSkillTest: async (opts: any) => { index++; current = scenarios[index]!; expect(current).toBeDefined(); calls.push(opts); expect(opts.testName).toBe(owner); expect(opts.maxTurns).toBe(8); expect(opts.timeout).toBe(CAPTURE_MS); - expect(opts.model).toBe('claude-sonnet-4-6'); expect(opts.runId).toBe('synthetic-posture-run'); + expect(opts.model).toBe(resolveEvalModel('capture')); expect(opts.runId).toBe('synthetic-posture-run'); expect(opts.signal).toBeInstanceOf(AbortSignal); expect(opts.prompt).toContain('Skip any AskUserQuestion'); const file = path.join(opts.workingDirectory, owner === owners[0] ? 'q3.md' : 'unlocks.md'); diff --git a/test/skill-e2e-auq-matrix.test.ts b/test/skill-e2e-auq-matrix.test.ts index a6717c2ef..1fe3bcee7 100644 --- a/test/skill-e2e-auq-matrix.test.ts +++ b/test/skill-e2e-auq-matrix.test.ts @@ -27,6 +27,7 @@ * Run a subset in the foreground with AUQ_MATRIX_ONLY="plan-eng-review,spec". */ import { test } from 'bun:test'; +import { resolveEvalModel } from '../lib/eval-model'; import { CAPTURE_MS } from './helpers/eval-budgets'; import { describeE2ETier } from './helpers/e2e-gate'; import * as fs from 'node:fs'; @@ -96,7 +97,7 @@ const MATRIX: MatrixSkill[] = [ // controlled Opus re-run passed cleanly (7/7 format, substance 5, 160s). // The spec workflow's long pre-question phase needs the stronger model // to reach its first AskUserQuestion inside the turn budget. - model: 'claude-opus-4-7', + model: resolveEvalModel('capture'), }, { skill: 'design-consultation', diff --git a/test/skill-e2e-office-hours-brain-writeback.test.ts b/test/skill-e2e-office-hours-brain-writeback.test.ts index fdeef14ee..12ec005f2 100644 --- a/test/skill-e2e-office-hours-brain-writeback.test.ts +++ b/test/skill-e2e-office-hours-brain-writeback.test.ts @@ -35,6 +35,7 @@ */ import { expect, beforeAll, afterAll } from 'bun:test'; +import { resolveEvalModel } from '../lib/eval-model'; import { CAPTURE_LONG_MS } from './helpers/eval-budgets'; import { execFileSync, spawnSync } from 'child_process'; import { @@ -228,7 +229,7 @@ exit 0 collector: evalCollector, name: '/office-hours-brain-writeback', suite: 'Office Hours Brain Writeback E2E', - model: 'claude-sonnet-4-6', + model: resolveEvalModel('capture'), run: (signal) => runSkillTest({ signal, prompt: `Read office-hours/SKILL.md for the workflow. @@ -245,7 +246,7 @@ This is a test of the brain-writeback path. Do NOT skip the gbrain save step und timeout: CAPTURE_LONG_MS, testName: 'office-hours-brain-writeback', runId, - model: 'claude-sonnet-4-6', + model: resolveEvalModel('capture'), env: childEnv, }), validate: (result) => { diff --git a/test/skill-e2e-office-hours.test.ts b/test/skill-e2e-office-hours.test.ts index e98f6d5a4..b9572a5e7 100644 --- a/test/skill-e2e-office-hours.test.ts +++ b/test/skill-e2e-office-hours.test.ts @@ -10,6 +10,7 @@ */ import { expect, beforeAll, afterAll } from 'bun:test'; +import { resolveEvalModel } from '../lib/eval-model'; import { CAPTURE_MS, CAPTURE_LONG_MS } from './helpers/eval-budgets'; import { runSkillTest } from './helpers/session-runner'; import { @@ -70,7 +71,7 @@ describeIfSelected('Office Hours Forcing Energy E2E', ['office-hours-forcing-ene judgeMetadata, name: '/office-hours-forcing-energy', suite: 'Office Hours Forcing Energy E2E', - model: 'claude-sonnet-4-6', + model: resolveEvalModel('capture'), run: (signal) => runSkillTest({ signal, prompt: `Read office-hours/SKILL.md for the workflow. @@ -85,7 +86,7 @@ Write Q3 output — the forcing question you would ask this founder — to ${wor timeout: CAPTURE_MS, testName: 'office-hours-forcing-energy', runId, - model: 'claude-sonnet-4-6', + model: resolveEvalModel('capture'), }), validate: async (result, signal) => { logCost('/office-hours (FORCING)', result); @@ -151,7 +152,7 @@ describeIfSelected('Office Hours Builder Wildness E2E', ['office-hours-builder-w judgeMetadata, name: '/office-hours-builder-wildness', suite: 'Office Hours Builder Wildness E2E', - model: 'claude-sonnet-4-6', + model: resolveEvalModel('capture'), run: (signal) => runSkillTest({ signal, prompt: `Read office-hours/SKILL.md for the workflow. @@ -166,7 +167,7 @@ Write your response — the three adjacent unlocks — to ${workDir}/unlocks.md. timeout: CAPTURE_MS, testName: 'office-hours-builder-wildness', runId, - model: 'claude-sonnet-4-6', + model: resolveEvalModel('capture'), }), validate: async (result, signal) => { logCost('/office-hours (BUILDER)', result); diff --git a/test/skill-e2e-plan-format.test.ts b/test/skill-e2e-plan-format.test.ts index 30981cf71..371365d66 100644 --- a/test/skill-e2e-plan-format.test.ts +++ b/test/skill-e2e-plan-format.test.ts @@ -18,6 +18,7 @@ * accordingly. */ import { describe, test, expect, beforeAll, afterAll } from 'bun:test'; +import { resolveEvalModel } from '../lib/eval-model'; import { CAPTURE_MS } from './helpers/eval-budgets'; import { runSkillTest } from './helpers/session-runner'; import { runRecordedOfficeHoursAttempt, OFFICE_HOURS_BUN_GRACE_MS } from './helpers/office-hours-attempt'; @@ -128,7 +129,7 @@ describeIfSelected('Plan Format — CEO Mode Selection', ['plan-ceo-review-forma const judgeMetadata: Pick = {}; await runRecordedOfficeHoursAttempt({ collector: evalCollector, name: '/plan-ceo-review-format-mode', suite: 'Plan Format — CEO Mode Selection', - model: 'claude-opus-4-7', budgetMs: CAPTURE_MS, + model: resolveEvalModel('capture'), budgetMs: CAPTURE_MS, judgeMetadata, run: signal => runSkillTest({ signal, @@ -146,7 +147,7 @@ After writing the file, stop. Do not continue the review.`, timeout: CAPTURE_MS, testName: 'plan-ceo-review-format-mode', runId, - model: 'claude-opus-4-7', + model: resolveEvalModel('capture'), }), validate: async (result, signal) => { logCost('/plan-ceo-review format (mode)', result); @@ -195,7 +196,7 @@ describeIfSelected('Plan Format — CEO Approach Menu', ['plan-ceo-review-format const judgeMetadata: Pick = {}; await runRecordedOfficeHoursAttempt({ collector: evalCollector, name: '/plan-ceo-review-format-approach', suite: 'Plan Format — CEO Approach Menu', - model: 'claude-opus-4-7', budgetMs: CAPTURE_MS, + model: resolveEvalModel('capture'), budgetMs: CAPTURE_MS, judgeMetadata, run: signal => runSkillTest({ signal, @@ -213,7 +214,7 @@ After writing the file, stop. Do not continue the review.`, timeout: CAPTURE_MS, testName: 'plan-ceo-review-format-approach', runId, - model: 'claude-opus-4-7', + model: resolveEvalModel('capture'), }), validate: async (result, signal) => { logCost('/plan-ceo-review format (approach)', result); @@ -261,7 +262,7 @@ describeIfSelected('Plan Format — Eng Coverage Issue', ['plan-eng-review-forma const judgeMetadata: Pick = {}; await runRecordedOfficeHoursAttempt({ collector: evalCollector, name: '/plan-eng-review-format-coverage', suite: 'Plan Format — Eng Coverage Issue', - model: 'claude-opus-4-7', budgetMs: CAPTURE_MS, + model: resolveEvalModel('capture'), budgetMs: CAPTURE_MS, judgeMetadata, run: signal => runSkillTest({ signal, @@ -282,7 +283,7 @@ After writing the file with that ONE question, stop. Do not continue the review. timeout: CAPTURE_MS, testName: 'plan-eng-review-format-coverage', runId, - model: 'claude-opus-4-7', + model: resolveEvalModel('capture'), }), validate: async (result, signal) => { logCost('/plan-eng-review format (coverage)', result); @@ -330,7 +331,7 @@ describeIfSelected('Plan Format — Eng Kind Issue', ['plan-eng-review-format-ki const judgeMetadata: Pick = {}; await runRecordedOfficeHoursAttempt({ collector: evalCollector, name: '/plan-eng-review-format-kind', suite: 'Plan Format — Eng Kind Issue', - model: 'claude-opus-4-7', budgetMs: CAPTURE_MS, + model: resolveEvalModel('capture'), budgetMs: CAPTURE_MS, judgeMetadata, run: signal => runSkillTest({ signal, @@ -348,7 +349,7 @@ After writing the file with that ONE question, stop. Do not continue the review. timeout: CAPTURE_MS, testName: 'plan-eng-review-format-kind', runId, - model: 'claude-opus-4-7', + model: resolveEvalModel('capture'), }), validate: async (result, signal) => { logCost('/plan-eng-review format (kind)', result); diff --git a/test/skill-e2e-qa-bugs.test.ts b/test/skill-e2e-qa-bugs.test.ts index 7858173d7..60024bff1 100644 --- a/test/skill-e2e-qa-bugs.test.ts +++ b/test/skill-e2e-qa-bugs.test.ts @@ -1,4 +1,5 @@ import { describe, test, expect, beforeAll, afterAll } from 'bun:test'; +import { resolveEvalModel } from '../lib/eval-model'; import { CAPTURE_MS, CAPTURE_LONG_MS } from './helpers/eval-budgets'; import { runSkillTest } from './helpers/session-runner'; import { outcomeJudge } from './helpers/llm-judge'; @@ -120,7 +121,7 @@ CRITICAL RULES: timeout: CAPTURE_MS, testName: `qa-${label}`, runId, - model: 'claude-opus-4-7', + model: resolveEvalModel('capture'), }); logCost(`/qa ${label}`, result); diff --git a/test/skill-e2e-retro.test.ts b/test/skill-e2e-retro.test.ts index e13a22d16..e4109e77d 100644 --- a/test/skill-e2e-retro.test.ts +++ b/test/skill-e2e-retro.test.ts @@ -1,4 +1,5 @@ import { expect, beforeAll, afterAll } from 'bun:test'; +import { resolveEvalModel } from '../lib/eval-model'; import { CAPTURE_MS, CAPTURE_LONG_MS } from './helpers/eval-budgets'; import { runSkillTest } from './helpers/session-runner'; import { @@ -202,7 +203,7 @@ Analyze the git history and produce the narrative report as described in the SKI timeout: CAPTURE_MS, testName: 'retro', runId, - model: 'claude-opus-4-7', + model: resolveEvalModel('capture'), }); logCost('/retro', result); diff --git a/test/skill-e2e-workflow.test.ts b/test/skill-e2e-workflow.test.ts index ca9d60624..9ffc14b76 100644 --- a/test/skill-e2e-workflow.test.ts +++ b/test/skill-e2e-workflow.test.ts @@ -1,4 +1,5 @@ import { describe, test, expect, beforeAll, afterAll } from 'bun:test'; +import { resolveEvalModel } from '../lib/eval-model'; import { JUDGE_MS, CAPTURE_MS, CAPTURE_LONG_MS } from './helpers/eval-budgets'; import { runSkillTest } from './helpers/session-runner'; import { @@ -16,7 +17,6 @@ import { extractSkillBody } from './helpers/skill-fixture'; import { createCoverageAuditFixture } from './fixtures/coverage-audit-fixture'; import { validateCoverageAudit, type CoverageFile } from './helpers/coverage-audit'; import { runRecordedOfficeHoursAttempt, OFFICE_HOURS_BUN_GRACE_MS } from './helpers/office-hours-attempt'; -import { resolveEvalModel } from '../lib/eval-model'; const evalCollector = createEvalCollector('e2e'); @@ -468,7 +468,7 @@ Write the full output (including the GATE verdict) to ${codexDir}/codex-output.m timeout: CAPTURE_MS, testName: 'codex-review', runId, - model: 'claude-opus-4-7', + model: resolveEvalModel('capture'), }); logCost('/codex review', result);