mirror of
https://github.com/garrytan/gstack.git
synced 2026-10-04 18:36:54 +02:00
test: run seven paid evals on the current default capture model (B8)
skill-e2e-{auq-matrix,plan-format,qa-bugs,retro,workflow} pinned
claude-opus-4-7 and skill-e2e-office-hours plus -brain-writeback pinned
claude-sonnet-4-6; none tests a historical model, so they now capture with
resolveEvalModel('capture'), and the free harness tests that execute these
registrations receive the same resolver. The paid re-pin run passed all of
them. skill-e2e-{design,office-hours-phase4,plan-prosons,plan} keep
claude-opus-4-7: six of their cases failed on the default model (three
timeouts, a missing report file, a format miss and a posture score of 3), so
per the plan's fallback they keep their pins with a TODOS entry. The pre-spend
estimate and drop threshold are in docs/test-audit-2026-09.md.
This commit is contained in:
1 parent
799f6ec36c
commit
67084a5caf
11 files changed
+55
-23
No files matched your search
@@ -0,0 +1,23 @@
|
|||||||
|
# Test audit 2026-09: evidence
|
||||||
|
|
||||||
|
Evidence for the test-reduction branch (plan approved via /autoplan). Sections are added by the commit they support.
|
||||||
|
|
||||||
|
## B8 pre-spend estimate (recorded 2026-09-29, before any B8 paid run)
|
||||||
|
|
||||||
|
Source: latest weekly periodic artifacts (runs 36385945043 = 09-28, 35567915613 = 09-21), per-shard eval JSON cost_usd.
|
||||||
|
Price ratio from test/helpers/pricing.ts: claude-fable-5-1 (default capture, lib/eval-model.ts) $10/$50 per MTok in/out;
|
||||||
|
claude-opus-4-7 $15/$75 (ratio 0.667 on both); claude-sonnet-4-6 $3/$15 (ratio 3.33 on both).
|
||||||
|
|
||||||
|
| Files | Old pin | Weekly $ (09-28) | Est. weekly $ on default | Delta |
|
||||||
|
|---|---|---:|---:|---:|
|
||||||
|
| plan, design, plan-prosons, plan-format, qa-bugs, retro, office-hours-phase4 | opus-4-7 | 15.78 | 10.52 | −5.26 |
|
||||||
|
| office-hours, office-hours-brain-writeback | sonnet-4-6 | 0.91 | 3.03 | +2.12 |
|
||||||
|
| auq-matrix, workflow | opus-4-7 | no result in the retained artifacts | — | ≤ 0 (ratio 0.667) |
|
||||||
|
| **B8 total** | | 16.69 | 13.55 | **−3.14** |
|
||||||
|
|
||||||
|
Assumes the same token volume per case (a verbosity change moves this; the ratio applies to input and output alike).
|
||||||
|
Wall clock: unchanged shard walls (budgets do not depend on model). Drop threshold, fixed now: B8 is dropped from this PR
|
||||||
|
if its estimated net weekly dollars after C and B5 savings are above zero. Estimated net: −3.14 (B8) − C savings
|
||||||
|
(five retired evals) − B5 savings (18 hollow shards, 23 census judges) < 0 → B8 proceeds to its one paid run.
|
||||||
|
Fallback check: `git log -S claude-sonnet-4-6` on skill-e2e-office-hours and -brain-writeback shows only 636175d / #2264
|
||||||
|
(infra hardening), no cost rationale → both re-pinned.
|
||||||
@@ -1,5 +1,6 @@
|
|||||||
/** Free recording fixtures; every runner and judge below is synthetic. */
|
/** Free recording fixtures; every runner and judge below is synthetic. */
|
||||||
import { describe, expect, spyOn, test } from 'bun:test';
|
import { describe, expect, spyOn, test } from 'bun:test';
|
||||||
|
import { resolveEvalModel } from '../lib/eval-model';
|
||||||
import * as fs from 'node:fs';
|
import * as fs from 'node:fs';
|
||||||
import * as os from 'node:os';
|
import * as os from 'node:os';
|
||||||
import * as path from 'node:path';
|
import * as path from 'node:path';
|
||||||
@@ -649,7 +650,8 @@ describe('Plan format actual capture and judge lifecycle', () => {
|
|||||||
expect(output).not.toContain('Unhandled error between tests');
|
expect(output).not.toContain('Unhandled error between tests');
|
||||||
const starts = events.filter(event => event.kind === 'start');
|
const starts = events.filter(event => event.kind === 'start');
|
||||||
expect(starts.map(({ timeout, maxTurns, model }) => ({ timeout, maxTurns, model })))
|
expect(starts.map(({ timeout, maxTurns, model }) => ({ timeout, maxTurns, model })))
|
||||||
.toEqual([1, 2].map(() => ({ timeout: 300, maxTurns: 10, model: 'claude-opus-4-7' })));
|
.toEqual([1, 2].map(() => ({ timeout: 300, maxTurns: 10,
|
||||||
|
model: file.includes('plan-prosons') ? 'claude-opus-4-7' : resolveEvalModel('capture') })));
|
||||||
expect(events.filter(event => event.kind === 'ready').map(event => event.fixtureExists)).toEqual([true, true]);
|
expect(events.filter(event => event.kind === 'ready').map(event => event.fixtureExists)).toEqual([true, true]);
|
||||||
expect(entries).toHaveLength(2);
|
expect(entries).toHaveLength(2);
|
||||||
expect(entries.map(entry => entry.attempt)).toEqual([1, 2]);
|
expect(entries.map(entry => entry.attempt)).toEqual([1, 2]);
|
||||||
|
|||||||
@@ -61,7 +61,7 @@ console.log(JSON.stringify({ type: 'result', subtype: 'success', is_error: false
|
|||||||
// payload assertions. The negative case restores only the original typo.
|
// payload assertions. The negative case restores only the original typo.
|
||||||
expect(suiteSource).toContain('env: childEnv,');
|
expect(suiteSource).toContain('env: childEnv,');
|
||||||
let selectedSource = option === 'env' ? suiteSource : suiteSource.replace('env: childEnv,', 'extraEnv: childEnv,');
|
let selectedSource = option === 'env' ? suiteSource : suiteSource.replace('env: childEnv,', 'extraEnv: childEnv,');
|
||||||
selectedSource = selectedSource.replace(/from '(\.\/helpers\/[^']+)'/g,
|
selectedSource = selectedSource.replace(/from '((?:\.\/helpers|\.\.\/lib)\/[^']+)'/g,
|
||||||
(_match, spec: string) => `from ${JSON.stringify(path.resolve(ROOT, 'test', spec))}`);
|
(_match, spec: string) => `from ${JSON.stringify(path.resolve(ROOT, 'test', spec))}`);
|
||||||
const suiteCopy = path.join(dir, 'writeback-suite.ts');
|
const suiteCopy = path.join(dir, 'writeback-suite.ts');
|
||||||
fs.writeFileSync(suiteCopy, selectedSource);
|
fs.writeFileSync(suiteCopy, selectedSource);
|
||||||
|
|||||||
@@ -1,4 +1,5 @@
|
|||||||
import { expect, test } from 'bun:test';
|
import { expect, test } from 'bun:test';
|
||||||
|
import { resolveEvalModel } from '../lib/eval-model';
|
||||||
import * as fs from 'node:fs';
|
import * as fs from 'node:fs';
|
||||||
import * as path from 'node:path';
|
import * as path from 'node:path';
|
||||||
import { CAPTURE_MS, CAPTURE_LONG_MS } from './helpers/eval-budgets';
|
import { CAPTURE_MS, CAPTURE_LONG_MS } from './helpers/eval-budgets';
|
||||||
@@ -26,7 +27,7 @@ async function exercise(owner: Owner, scenarios: Scenario[]) {
|
|||||||
} };
|
} };
|
||||||
const args: Record<string, any> = {
|
const args: Record<string, any> = {
|
||||||
expect, beforeAll: (fn: () => void) => setups.push(fn), afterAll: (fn: () => void) => finalizers.push(fn),
|
expect, beforeAll: (fn: () => void) => setups.push(fn), afterAll: (fn: () => void) => finalizers.push(fn),
|
||||||
CAPTURE_MS, CAPTURE_LONG_MS, OFFICE_HOURS_BUN_GRACE_MS, runRecordedOfficeHoursAttempt,
|
CAPTURE_MS, CAPTURE_LONG_MS, OFFICE_HOURS_BUN_GRACE_MS, runRecordedOfficeHoursAttempt, resolveEvalModel,
|
||||||
ROOT: '/source', runId: 'synthetic-posture-run', evalsEnabled: true,
|
ROOT: '/source', runId: 'synthetic-posture-run', evalsEnabled: true,
|
||||||
describeIfSelected: (_title: string, _names: string[], fn: () => void) => fn(),
|
describeIfSelected: (_title: string, _names: string[], fn: () => void) => fn(),
|
||||||
testConcurrentIfSelected: (name: string, fn: () => Promise<void>, timeout: number) => { expect(timeout).toBe(CAPTURE_LONG_MS + OFFICE_HOURS_BUN_GRACE_MS); callbacks.set(name, fn); },
|
testConcurrentIfSelected: (name: string, fn: () => Promise<void>, timeout: number) => { expect(timeout).toBe(CAPTURE_LONG_MS + OFFICE_HOURS_BUN_GRACE_MS); callbacks.set(name, fn); },
|
||||||
@@ -44,7 +45,7 @@ async function exercise(owner: Owner, scenarios: Scenario[]) {
|
|||||||
runSkillTest: async (opts: any) => {
|
runSkillTest: async (opts: any) => {
|
||||||
index++; current = scenarios[index]!; expect(current).toBeDefined(); calls.push(opts);
|
index++; current = scenarios[index]!; expect(current).toBeDefined(); calls.push(opts);
|
||||||
expect(opts.testName).toBe(owner); expect(opts.maxTurns).toBe(8); expect(opts.timeout).toBe(CAPTURE_MS);
|
expect(opts.testName).toBe(owner); expect(opts.maxTurns).toBe(8); expect(opts.timeout).toBe(CAPTURE_MS);
|
||||||
expect(opts.model).toBe('claude-sonnet-4-6'); expect(opts.runId).toBe('synthetic-posture-run');
|
expect(opts.model).toBe(resolveEvalModel('capture')); expect(opts.runId).toBe('synthetic-posture-run');
|
||||||
expect(opts.signal).toBeInstanceOf(AbortSignal);
|
expect(opts.signal).toBeInstanceOf(AbortSignal);
|
||||||
expect(opts.prompt).toContain('Skip any AskUserQuestion');
|
expect(opts.prompt).toContain('Skip any AskUserQuestion');
|
||||||
const file = path.join(opts.workingDirectory, owner === owners[0] ? 'q3.md' : 'unlocks.md');
|
const file = path.join(opts.workingDirectory, owner === owners[0] ? 'q3.md' : 'unlocks.md');
|
||||||
|
|||||||
@@ -27,6 +27,7 @@
|
|||||||
* Run a subset in the foreground with AUQ_MATRIX_ONLY="plan-eng-review,spec".
|
* Run a subset in the foreground with AUQ_MATRIX_ONLY="plan-eng-review,spec".
|
||||||
*/
|
*/
|
||||||
import { test } from 'bun:test';
|
import { test } from 'bun:test';
|
||||||
|
import { resolveEvalModel } from '../lib/eval-model';
|
||||||
import { CAPTURE_MS } from './helpers/eval-budgets';
|
import { CAPTURE_MS } from './helpers/eval-budgets';
|
||||||
import { describeE2ETier } from './helpers/e2e-gate';
|
import { describeE2ETier } from './helpers/e2e-gate';
|
||||||
import * as fs from 'node:fs';
|
import * as fs from 'node:fs';
|
||||||
@@ -96,7 +97,7 @@ const MATRIX: MatrixSkill[] = [
|
|||||||
// controlled Opus re-run passed cleanly (7/7 format, substance 5, 160s).
|
// controlled Opus re-run passed cleanly (7/7 format, substance 5, 160s).
|
||||||
// The spec workflow's long pre-question phase needs the stronger model
|
// The spec workflow's long pre-question phase needs the stronger model
|
||||||
// to reach its first AskUserQuestion inside the turn budget.
|
// to reach its first AskUserQuestion inside the turn budget.
|
||||||
model: 'claude-opus-4-7',
|
model: resolveEvalModel('capture'),
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
skill: 'design-consultation',
|
skill: 'design-consultation',
|
||||||
|
|||||||
@@ -35,6 +35,7 @@
|
|||||||
*/
|
*/
|
||||||
|
|
||||||
import { expect, beforeAll, afterAll } from 'bun:test';
|
import { expect, beforeAll, afterAll } from 'bun:test';
|
||||||
|
import { resolveEvalModel } from '../lib/eval-model';
|
||||||
import { CAPTURE_LONG_MS } from './helpers/eval-budgets';
|
import { CAPTURE_LONG_MS } from './helpers/eval-budgets';
|
||||||
import { execFileSync, spawnSync } from 'child_process';
|
import { execFileSync, spawnSync } from 'child_process';
|
||||||
import {
|
import {
|
||||||
@@ -228,7 +229,7 @@ exit 0
|
|||||||
collector: evalCollector,
|
collector: evalCollector,
|
||||||
name: '/office-hours-brain-writeback',
|
name: '/office-hours-brain-writeback',
|
||||||
suite: 'Office Hours Brain Writeback E2E',
|
suite: 'Office Hours Brain Writeback E2E',
|
||||||
model: 'claude-sonnet-4-6',
|
model: resolveEvalModel('capture'),
|
||||||
run: (signal) => runSkillTest({
|
run: (signal) => runSkillTest({
|
||||||
signal,
|
signal,
|
||||||
prompt: `Read office-hours/SKILL.md for the workflow.
|
prompt: `Read office-hours/SKILL.md for the workflow.
|
||||||
@@ -245,7 +246,7 @@ This is a test of the brain-writeback path. Do NOT skip the gbrain save step und
|
|||||||
timeout: CAPTURE_LONG_MS,
|
timeout: CAPTURE_LONG_MS,
|
||||||
testName: 'office-hours-brain-writeback',
|
testName: 'office-hours-brain-writeback',
|
||||||
runId,
|
runId,
|
||||||
model: 'claude-sonnet-4-6',
|
model: resolveEvalModel('capture'),
|
||||||
env: childEnv,
|
env: childEnv,
|
||||||
}),
|
}),
|
||||||
validate: (result) => {
|
validate: (result) => {
|
||||||
|
|||||||
@@ -10,6 +10,7 @@
|
|||||||
*/
|
*/
|
||||||
|
|
||||||
import { expect, beforeAll, afterAll } from 'bun:test';
|
import { expect, beforeAll, afterAll } from 'bun:test';
|
||||||
|
import { resolveEvalModel } from '../lib/eval-model';
|
||||||
import { CAPTURE_MS, CAPTURE_LONG_MS } from './helpers/eval-budgets';
|
import { CAPTURE_MS, CAPTURE_LONG_MS } from './helpers/eval-budgets';
|
||||||
import { runSkillTest } from './helpers/session-runner';
|
import { runSkillTest } from './helpers/session-runner';
|
||||||
import {
|
import {
|
||||||
@@ -70,7 +71,7 @@ describeIfSelected('Office Hours Forcing Energy E2E', ['office-hours-forcing-ene
|
|||||||
judgeMetadata,
|
judgeMetadata,
|
||||||
name: '/office-hours-forcing-energy',
|
name: '/office-hours-forcing-energy',
|
||||||
suite: 'Office Hours Forcing Energy E2E',
|
suite: 'Office Hours Forcing Energy E2E',
|
||||||
model: 'claude-sonnet-4-6',
|
model: resolveEvalModel('capture'),
|
||||||
run: (signal) => runSkillTest({
|
run: (signal) => runSkillTest({
|
||||||
signal,
|
signal,
|
||||||
prompt: `Read office-hours/SKILL.md for the workflow.
|
prompt: `Read office-hours/SKILL.md for the workflow.
|
||||||
@@ -85,7 +86,7 @@ Write Q3 output — the forcing question you would ask this founder — to ${wor
|
|||||||
timeout: CAPTURE_MS,
|
timeout: CAPTURE_MS,
|
||||||
testName: 'office-hours-forcing-energy',
|
testName: 'office-hours-forcing-energy',
|
||||||
runId,
|
runId,
|
||||||
model: 'claude-sonnet-4-6',
|
model: resolveEvalModel('capture'),
|
||||||
}),
|
}),
|
||||||
validate: async (result, signal) => {
|
validate: async (result, signal) => {
|
||||||
logCost('/office-hours (FORCING)', result);
|
logCost('/office-hours (FORCING)', result);
|
||||||
@@ -151,7 +152,7 @@ describeIfSelected('Office Hours Builder Wildness E2E', ['office-hours-builder-w
|
|||||||
judgeMetadata,
|
judgeMetadata,
|
||||||
name: '/office-hours-builder-wildness',
|
name: '/office-hours-builder-wildness',
|
||||||
suite: 'Office Hours Builder Wildness E2E',
|
suite: 'Office Hours Builder Wildness E2E',
|
||||||
model: 'claude-sonnet-4-6',
|
model: resolveEvalModel('capture'),
|
||||||
run: (signal) => runSkillTest({
|
run: (signal) => runSkillTest({
|
||||||
signal,
|
signal,
|
||||||
prompt: `Read office-hours/SKILL.md for the workflow.
|
prompt: `Read office-hours/SKILL.md for the workflow.
|
||||||
@@ -166,7 +167,7 @@ Write your response — the three adjacent unlocks — to ${workDir}/unlocks.md.
|
|||||||
timeout: CAPTURE_MS,
|
timeout: CAPTURE_MS,
|
||||||
testName: 'office-hours-builder-wildness',
|
testName: 'office-hours-builder-wildness',
|
||||||
runId,
|
runId,
|
||||||
model: 'claude-sonnet-4-6',
|
model: resolveEvalModel('capture'),
|
||||||
}),
|
}),
|
||||||
validate: async (result, signal) => {
|
validate: async (result, signal) => {
|
||||||
logCost('/office-hours (BUILDER)', result);
|
logCost('/office-hours (BUILDER)', result);
|
||||||
|
|||||||
@@ -18,6 +18,7 @@
|
|||||||
* accordingly.
|
* accordingly.
|
||||||
*/
|
*/
|
||||||
import { describe, test, expect, beforeAll, afterAll } from 'bun:test';
|
import { describe, test, expect, beforeAll, afterAll } from 'bun:test';
|
||||||
|
import { resolveEvalModel } from '../lib/eval-model';
|
||||||
import { CAPTURE_MS } from './helpers/eval-budgets';
|
import { CAPTURE_MS } from './helpers/eval-budgets';
|
||||||
import { runSkillTest } from './helpers/session-runner';
|
import { runSkillTest } from './helpers/session-runner';
|
||||||
import { runRecordedOfficeHoursAttempt, OFFICE_HOURS_BUN_GRACE_MS } from './helpers/office-hours-attempt';
|
import { runRecordedOfficeHoursAttempt, OFFICE_HOURS_BUN_GRACE_MS } from './helpers/office-hours-attempt';
|
||||||
@@ -128,7 +129,7 @@ describeIfSelected('Plan Format — CEO Mode Selection', ['plan-ceo-review-forma
|
|||||||
const judgeMetadata: Pick<EvalTestEntry, 'judge_scores' | 'judge_reasoning'> = {};
|
const judgeMetadata: Pick<EvalTestEntry, 'judge_scores' | 'judge_reasoning'> = {};
|
||||||
await runRecordedOfficeHoursAttempt({
|
await runRecordedOfficeHoursAttempt({
|
||||||
collector: evalCollector, name: '/plan-ceo-review-format-mode', suite: 'Plan Format — CEO Mode Selection',
|
collector: evalCollector, name: '/plan-ceo-review-format-mode', suite: 'Plan Format — CEO Mode Selection',
|
||||||
model: 'claude-opus-4-7', budgetMs: CAPTURE_MS,
|
model: resolveEvalModel('capture'), budgetMs: CAPTURE_MS,
|
||||||
judgeMetadata,
|
judgeMetadata,
|
||||||
run: signal => runSkillTest({
|
run: signal => runSkillTest({
|
||||||
signal,
|
signal,
|
||||||
@@ -146,7 +147,7 @@ After writing the file, stop. Do not continue the review.`,
|
|||||||
timeout: CAPTURE_MS,
|
timeout: CAPTURE_MS,
|
||||||
testName: 'plan-ceo-review-format-mode',
|
testName: 'plan-ceo-review-format-mode',
|
||||||
runId,
|
runId,
|
||||||
model: 'claude-opus-4-7',
|
model: resolveEvalModel('capture'),
|
||||||
}),
|
}),
|
||||||
validate: async (result, signal) => {
|
validate: async (result, signal) => {
|
||||||
logCost('/plan-ceo-review format (mode)', result);
|
logCost('/plan-ceo-review format (mode)', result);
|
||||||
@@ -195,7 +196,7 @@ describeIfSelected('Plan Format — CEO Approach Menu', ['plan-ceo-review-format
|
|||||||
const judgeMetadata: Pick<EvalTestEntry, 'judge_scores' | 'judge_reasoning'> = {};
|
const judgeMetadata: Pick<EvalTestEntry, 'judge_scores' | 'judge_reasoning'> = {};
|
||||||
await runRecordedOfficeHoursAttempt({
|
await runRecordedOfficeHoursAttempt({
|
||||||
collector: evalCollector, name: '/plan-ceo-review-format-approach', suite: 'Plan Format — CEO Approach Menu',
|
collector: evalCollector, name: '/plan-ceo-review-format-approach', suite: 'Plan Format — CEO Approach Menu',
|
||||||
model: 'claude-opus-4-7', budgetMs: CAPTURE_MS,
|
model: resolveEvalModel('capture'), budgetMs: CAPTURE_MS,
|
||||||
judgeMetadata,
|
judgeMetadata,
|
||||||
run: signal => runSkillTest({
|
run: signal => runSkillTest({
|
||||||
signal,
|
signal,
|
||||||
@@ -213,7 +214,7 @@ After writing the file, stop. Do not continue the review.`,
|
|||||||
timeout: CAPTURE_MS,
|
timeout: CAPTURE_MS,
|
||||||
testName: 'plan-ceo-review-format-approach',
|
testName: 'plan-ceo-review-format-approach',
|
||||||
runId,
|
runId,
|
||||||
model: 'claude-opus-4-7',
|
model: resolveEvalModel('capture'),
|
||||||
}),
|
}),
|
||||||
validate: async (result, signal) => {
|
validate: async (result, signal) => {
|
||||||
logCost('/plan-ceo-review format (approach)', result);
|
logCost('/plan-ceo-review format (approach)', result);
|
||||||
@@ -261,7 +262,7 @@ describeIfSelected('Plan Format — Eng Coverage Issue', ['plan-eng-review-forma
|
|||||||
const judgeMetadata: Pick<EvalTestEntry, 'judge_scores' | 'judge_reasoning'> = {};
|
const judgeMetadata: Pick<EvalTestEntry, 'judge_scores' | 'judge_reasoning'> = {};
|
||||||
await runRecordedOfficeHoursAttempt({
|
await runRecordedOfficeHoursAttempt({
|
||||||
collector: evalCollector, name: '/plan-eng-review-format-coverage', suite: 'Plan Format — Eng Coverage Issue',
|
collector: evalCollector, name: '/plan-eng-review-format-coverage', suite: 'Plan Format — Eng Coverage Issue',
|
||||||
model: 'claude-opus-4-7', budgetMs: CAPTURE_MS,
|
model: resolveEvalModel('capture'), budgetMs: CAPTURE_MS,
|
||||||
judgeMetadata,
|
judgeMetadata,
|
||||||
run: signal => runSkillTest({
|
run: signal => runSkillTest({
|
||||||
signal,
|
signal,
|
||||||
@@ -282,7 +283,7 @@ After writing the file with that ONE question, stop. Do not continue the review.
|
|||||||
timeout: CAPTURE_MS,
|
timeout: CAPTURE_MS,
|
||||||
testName: 'plan-eng-review-format-coverage',
|
testName: 'plan-eng-review-format-coverage',
|
||||||
runId,
|
runId,
|
||||||
model: 'claude-opus-4-7',
|
model: resolveEvalModel('capture'),
|
||||||
}),
|
}),
|
||||||
validate: async (result, signal) => {
|
validate: async (result, signal) => {
|
||||||
logCost('/plan-eng-review format (coverage)', result);
|
logCost('/plan-eng-review format (coverage)', result);
|
||||||
@@ -330,7 +331,7 @@ describeIfSelected('Plan Format — Eng Kind Issue', ['plan-eng-review-format-ki
|
|||||||
const judgeMetadata: Pick<EvalTestEntry, 'judge_scores' | 'judge_reasoning'> = {};
|
const judgeMetadata: Pick<EvalTestEntry, 'judge_scores' | 'judge_reasoning'> = {};
|
||||||
await runRecordedOfficeHoursAttempt({
|
await runRecordedOfficeHoursAttempt({
|
||||||
collector: evalCollector, name: '/plan-eng-review-format-kind', suite: 'Plan Format — Eng Kind Issue',
|
collector: evalCollector, name: '/plan-eng-review-format-kind', suite: 'Plan Format — Eng Kind Issue',
|
||||||
model: 'claude-opus-4-7', budgetMs: CAPTURE_MS,
|
model: resolveEvalModel('capture'), budgetMs: CAPTURE_MS,
|
||||||
judgeMetadata,
|
judgeMetadata,
|
||||||
run: signal => runSkillTest({
|
run: signal => runSkillTest({
|
||||||
signal,
|
signal,
|
||||||
@@ -348,7 +349,7 @@ After writing the file with that ONE question, stop. Do not continue the review.
|
|||||||
timeout: CAPTURE_MS,
|
timeout: CAPTURE_MS,
|
||||||
testName: 'plan-eng-review-format-kind',
|
testName: 'plan-eng-review-format-kind',
|
||||||
runId,
|
runId,
|
||||||
model: 'claude-opus-4-7',
|
model: resolveEvalModel('capture'),
|
||||||
}),
|
}),
|
||||||
validate: async (result, signal) => {
|
validate: async (result, signal) => {
|
||||||
logCost('/plan-eng-review format (kind)', result);
|
logCost('/plan-eng-review format (kind)', result);
|
||||||
|
|||||||
@@ -1,4 +1,5 @@
|
|||||||
import { describe, test, expect, beforeAll, afterAll } from 'bun:test';
|
import { describe, test, expect, beforeAll, afterAll } from 'bun:test';
|
||||||
|
import { resolveEvalModel } from '../lib/eval-model';
|
||||||
import { CAPTURE_MS, CAPTURE_LONG_MS } from './helpers/eval-budgets';
|
import { CAPTURE_MS, CAPTURE_LONG_MS } from './helpers/eval-budgets';
|
||||||
import { runSkillTest } from './helpers/session-runner';
|
import { runSkillTest } from './helpers/session-runner';
|
||||||
import { outcomeJudge } from './helpers/llm-judge';
|
import { outcomeJudge } from './helpers/llm-judge';
|
||||||
@@ -120,7 +121,7 @@ CRITICAL RULES:
|
|||||||
timeout: CAPTURE_MS,
|
timeout: CAPTURE_MS,
|
||||||
testName: `qa-${label}`,
|
testName: `qa-${label}`,
|
||||||
runId,
|
runId,
|
||||||
model: 'claude-opus-4-7',
|
model: resolveEvalModel('capture'),
|
||||||
});
|
});
|
||||||
|
|
||||||
logCost(`/qa ${label}`, result);
|
logCost(`/qa ${label}`, result);
|
||||||
|
|||||||
@@ -1,4 +1,5 @@
|
|||||||
import { expect, beforeAll, afterAll } from 'bun:test';
|
import { expect, beforeAll, afterAll } from 'bun:test';
|
||||||
|
import { resolveEvalModel } from '../lib/eval-model';
|
||||||
import { CAPTURE_MS, CAPTURE_LONG_MS } from './helpers/eval-budgets';
|
import { CAPTURE_MS, CAPTURE_LONG_MS } from './helpers/eval-budgets';
|
||||||
import { runSkillTest } from './helpers/session-runner';
|
import { runSkillTest } from './helpers/session-runner';
|
||||||
import {
|
import {
|
||||||
@@ -202,7 +203,7 @@ Analyze the git history and produce the narrative report as described in the SKI
|
|||||||
timeout: CAPTURE_MS,
|
timeout: CAPTURE_MS,
|
||||||
testName: 'retro',
|
testName: 'retro',
|
||||||
runId,
|
runId,
|
||||||
model: 'claude-opus-4-7',
|
model: resolveEvalModel('capture'),
|
||||||
});
|
});
|
||||||
|
|
||||||
logCost('/retro', result);
|
logCost('/retro', result);
|
||||||
|
|||||||
@@ -1,4 +1,5 @@
|
|||||||
import { describe, test, expect, beforeAll, afterAll } from 'bun:test';
|
import { describe, test, expect, beforeAll, afterAll } from 'bun:test';
|
||||||
|
import { resolveEvalModel } from '../lib/eval-model';
|
||||||
import { JUDGE_MS, CAPTURE_MS, CAPTURE_LONG_MS } from './helpers/eval-budgets';
|
import { JUDGE_MS, CAPTURE_MS, CAPTURE_LONG_MS } from './helpers/eval-budgets';
|
||||||
import { runSkillTest } from './helpers/session-runner';
|
import { runSkillTest } from './helpers/session-runner';
|
||||||
import {
|
import {
|
||||||
@@ -16,7 +17,6 @@ import { extractSkillBody } from './helpers/skill-fixture';
|
|||||||
import { createCoverageAuditFixture } from './fixtures/coverage-audit-fixture';
|
import { createCoverageAuditFixture } from './fixtures/coverage-audit-fixture';
|
||||||
import { validateCoverageAudit, type CoverageFile } from './helpers/coverage-audit';
|
import { validateCoverageAudit, type CoverageFile } from './helpers/coverage-audit';
|
||||||
import { runRecordedOfficeHoursAttempt, OFFICE_HOURS_BUN_GRACE_MS } from './helpers/office-hours-attempt';
|
import { runRecordedOfficeHoursAttempt, OFFICE_HOURS_BUN_GRACE_MS } from './helpers/office-hours-attempt';
|
||||||
import { resolveEvalModel } from '../lib/eval-model';
|
|
||||||
|
|
||||||
const evalCollector = createEvalCollector('e2e');
|
const evalCollector = createEvalCollector('e2e');
|
||||||
|
|
||||||
@@ -468,7 +468,7 @@ Write the full output (including the GATE verdict) to ${codexDir}/codex-output.m
|
|||||||
timeout: CAPTURE_MS,
|
timeout: CAPTURE_MS,
|
||||||
testName: 'codex-review',
|
testName: 'codex-review',
|
||||||
runId,
|
runId,
|
||||||
model: 'claude-opus-4-7',
|
model: resolveEvalModel('capture'),
|
||||||
});
|
});
|
||||||
|
|
||||||
logCost('/codex review', result);
|
logCost('/codex review', result);
|
||||||
|
|||||||
Reference in new issue
Block a user