mirror of
https://github.com/garrytan/gstack.git
synced 2026-10-03 09:56:57 +02:00
test: run seven paid evals on the current default capture model (B8)
skill-e2e-{auq-matrix,plan-format,qa-bugs,retro,workflow} pinned
claude-opus-4-7 and skill-e2e-office-hours plus -brain-writeback pinned
claude-sonnet-4-6; none tests a historical model, so they now capture with
resolveEvalModel('capture'), and the free harness tests that execute these
registrations receive the same resolver. The paid re-pin run passed all of
them. skill-e2e-{design,office-hours-phase4,plan-prosons,plan} keep
claude-opus-4-7: six of their cases failed on the default model (three
timeouts, a missing report file, a format miss and a posture score of 3), so
per the plan's fallback they keep their pins with a TODOS entry. The pre-spend
estimate and drop threshold are in docs/test-audit-2026-09.md.
This commit is contained in:
1 parent
799f6ec36c
commit
67084a5caf
11 files changed
+55
-23
No files matched your search
@@ -18,6 +18,7 @@
|
||||
* accordingly.
|
||||
*/
|
||||
import { describe, test, expect, beforeAll, afterAll } from 'bun:test';
|
||||
import { resolveEvalModel } from '../lib/eval-model';
|
||||
import { CAPTURE_MS } from './helpers/eval-budgets';
|
||||
import { runSkillTest } from './helpers/session-runner';
|
||||
import { runRecordedOfficeHoursAttempt, OFFICE_HOURS_BUN_GRACE_MS } from './helpers/office-hours-attempt';
|
||||
@@ -128,7 +129,7 @@ describeIfSelected('Plan Format — CEO Mode Selection', ['plan-ceo-review-forma
|
||||
const judgeMetadata: Pick<EvalTestEntry, 'judge_scores' | 'judge_reasoning'> = {};
|
||||
await runRecordedOfficeHoursAttempt({
|
||||
collector: evalCollector, name: '/plan-ceo-review-format-mode', suite: 'Plan Format — CEO Mode Selection',
|
||||
model: 'claude-opus-4-7', budgetMs: CAPTURE_MS,
|
||||
model: resolveEvalModel('capture'), budgetMs: CAPTURE_MS,
|
||||
judgeMetadata,
|
||||
run: signal => runSkillTest({
|
||||
signal,
|
||||
@@ -146,7 +147,7 @@ After writing the file, stop. Do not continue the review.`,
|
||||
timeout: CAPTURE_MS,
|
||||
testName: 'plan-ceo-review-format-mode',
|
||||
runId,
|
||||
model: 'claude-opus-4-7',
|
||||
model: resolveEvalModel('capture'),
|
||||
}),
|
||||
validate: async (result, signal) => {
|
||||
logCost('/plan-ceo-review format (mode)', result);
|
||||
@@ -195,7 +196,7 @@ describeIfSelected('Plan Format — CEO Approach Menu', ['plan-ceo-review-format
|
||||
const judgeMetadata: Pick<EvalTestEntry, 'judge_scores' | 'judge_reasoning'> = {};
|
||||
await runRecordedOfficeHoursAttempt({
|
||||
collector: evalCollector, name: '/plan-ceo-review-format-approach', suite: 'Plan Format — CEO Approach Menu',
|
||||
model: 'claude-opus-4-7', budgetMs: CAPTURE_MS,
|
||||
model: resolveEvalModel('capture'), budgetMs: CAPTURE_MS,
|
||||
judgeMetadata,
|
||||
run: signal => runSkillTest({
|
||||
signal,
|
||||
@@ -213,7 +214,7 @@ After writing the file, stop. Do not continue the review.`,
|
||||
timeout: CAPTURE_MS,
|
||||
testName: 'plan-ceo-review-format-approach',
|
||||
runId,
|
||||
model: 'claude-opus-4-7',
|
||||
model: resolveEvalModel('capture'),
|
||||
}),
|
||||
validate: async (result, signal) => {
|
||||
logCost('/plan-ceo-review format (approach)', result);
|
||||
@@ -261,7 +262,7 @@ describeIfSelected('Plan Format — Eng Coverage Issue', ['plan-eng-review-forma
|
||||
const judgeMetadata: Pick<EvalTestEntry, 'judge_scores' | 'judge_reasoning'> = {};
|
||||
await runRecordedOfficeHoursAttempt({
|
||||
collector: evalCollector, name: '/plan-eng-review-format-coverage', suite: 'Plan Format — Eng Coverage Issue',
|
||||
model: 'claude-opus-4-7', budgetMs: CAPTURE_MS,
|
||||
model: resolveEvalModel('capture'), budgetMs: CAPTURE_MS,
|
||||
judgeMetadata,
|
||||
run: signal => runSkillTest({
|
||||
signal,
|
||||
@@ -282,7 +283,7 @@ After writing the file with that ONE question, stop. Do not continue the review.
|
||||
timeout: CAPTURE_MS,
|
||||
testName: 'plan-eng-review-format-coverage',
|
||||
runId,
|
||||
model: 'claude-opus-4-7',
|
||||
model: resolveEvalModel('capture'),
|
||||
}),
|
||||
validate: async (result, signal) => {
|
||||
logCost('/plan-eng-review format (coverage)', result);
|
||||
@@ -330,7 +331,7 @@ describeIfSelected('Plan Format — Eng Kind Issue', ['plan-eng-review-format-ki
|
||||
const judgeMetadata: Pick<EvalTestEntry, 'judge_scores' | 'judge_reasoning'> = {};
|
||||
await runRecordedOfficeHoursAttempt({
|
||||
collector: evalCollector, name: '/plan-eng-review-format-kind', suite: 'Plan Format — Eng Kind Issue',
|
||||
model: 'claude-opus-4-7', budgetMs: CAPTURE_MS,
|
||||
model: resolveEvalModel('capture'), budgetMs: CAPTURE_MS,
|
||||
judgeMetadata,
|
||||
run: signal => runSkillTest({
|
||||
signal,
|
||||
@@ -348,7 +349,7 @@ After writing the file with that ONE question, stop. Do not continue the review.
|
||||
timeout: CAPTURE_MS,
|
||||
testName: 'plan-eng-review-format-kind',
|
||||
runId,
|
||||
model: 'claude-opus-4-7',
|
||||
model: resolveEvalModel('capture'),
|
||||
}),
|
||||
validate: async (result, signal) => {
|
||||
logCost('/plan-eng-review format (kind)', result);
|
||||
|
||||
Reference in new issue
Block a user