mirror of
https://github.com/garrytan/gstack.git
synced 2026-10-02 17:40:02 +02:00
test: run seven paid evals on the current default capture model (B8)
skill-e2e-{auq-matrix,plan-format,qa-bugs,retro,workflow} pinned
claude-opus-4-7 and skill-e2e-office-hours plus -brain-writeback pinned
claude-sonnet-4-6; none tests a historical model, so they now capture with
resolveEvalModel('capture'), and the free harness tests that execute these
registrations receive the same resolver. The paid re-pin run passed all of
them. skill-e2e-{design,office-hours-phase4,plan-prosons,plan} keep
claude-opus-4-7: six of their cases failed on the default model (three
timeouts, a missing report file, a format miss and a posture score of 3), so
per the plan's fallback they keep their pins with a TODOS entry. The pre-spend
estimate and drop threshold are in docs/test-audit-2026-09.md.
This commit is contained in:
1 parent
799f6ec36c
commit
67084a5caf
11 files changed
+55
-23
No files matched your search
@@ -10,6 +10,7 @@
|
||||
*/
|
||||
|
||||
import { expect, beforeAll, afterAll } from 'bun:test';
|
||||
import { resolveEvalModel } from '../lib/eval-model';
|
||||
import { CAPTURE_MS, CAPTURE_LONG_MS } from './helpers/eval-budgets';
|
||||
import { runSkillTest } from './helpers/session-runner';
|
||||
import {
|
||||
@@ -70,7 +71,7 @@ describeIfSelected('Office Hours Forcing Energy E2E', ['office-hours-forcing-ene
|
||||
judgeMetadata,
|
||||
name: '/office-hours-forcing-energy',
|
||||
suite: 'Office Hours Forcing Energy E2E',
|
||||
model: 'claude-sonnet-4-6',
|
||||
model: resolveEvalModel('capture'),
|
||||
run: (signal) => runSkillTest({
|
||||
signal,
|
||||
prompt: `Read office-hours/SKILL.md for the workflow.
|
||||
@@ -85,7 +86,7 @@ Write Q3 output — the forcing question you would ask this founder — to ${wor
|
||||
timeout: CAPTURE_MS,
|
||||
testName: 'office-hours-forcing-energy',
|
||||
runId,
|
||||
model: 'claude-sonnet-4-6',
|
||||
model: resolveEvalModel('capture'),
|
||||
}),
|
||||
validate: async (result, signal) => {
|
||||
logCost('/office-hours (FORCING)', result);
|
||||
@@ -151,7 +152,7 @@ describeIfSelected('Office Hours Builder Wildness E2E', ['office-hours-builder-w
|
||||
judgeMetadata,
|
||||
name: '/office-hours-builder-wildness',
|
||||
suite: 'Office Hours Builder Wildness E2E',
|
||||
model: 'claude-sonnet-4-6',
|
||||
model: resolveEvalModel('capture'),
|
||||
run: (signal) => runSkillTest({
|
||||
signal,
|
||||
prompt: `Read office-hours/SKILL.md for the workflow.
|
||||
@@ -166,7 +167,7 @@ Write your response — the three adjacent unlocks — to ${workDir}/unlocks.md.
|
||||
timeout: CAPTURE_MS,
|
||||
testName: 'office-hours-builder-wildness',
|
||||
runId,
|
||||
model: 'claude-sonnet-4-6',
|
||||
model: resolveEvalModel('capture'),
|
||||
}),
|
||||
validate: async (result, signal) => {
|
||||
logCost('/office-hours (BUILDER)', result);
|
||||
|
||||
Reference in new issue
Block a user