feat(evals): with-skill vs without-skill arm benchmark — measures whether gstack's behavioral layer earns its tokens

Ponytail's honest-benchmark method pointed at gstack itself: 3 build-shaped
tasks (native-platform over-build trap, CRUD endpoint, bug fix with planted
decoys) x 2 arms, real claude -p sessions, scored on the git diff left
behind. A research instrument, not a release gate — no assertion compares
arm scores.

Arms use the PROVEN project-scope pattern: the with-arm installs a
build-discipline skill (extracted reuse-ladder + bounded-closer content, not
whole-file copies) into the fixture's .claude/skills/ with a CLAUDE.md
routing line and an explicit invocation; a live spike confirmed claude -p
discovers and invokes project-scope skills via the Skill tool (3 turns,
exact-output probe). Fixtures are git init + local bare origin; diff capture
is three lines of git, no worktree machinery.

Failure taxonomy: zero-diff arms are VALID scored cells (deterministic
0/none, no API call), harvest failures record harvest:null, judge_error
cells are excluded from aggregates but named in the report — nothing drops
silently. armJudge: fixed sonnet judge, 0-3 unrequested-structure rubric,
must name the construct or say none, bounded retry-on-malformed; callJudge
gains optional temperature/max_tokens (defaults unchanged). recordE2E now
populates tokens_used for every E2E. Eval schema v2: harvest gains
{insertions, deletions, net}, tolerant reads keep v1 runs comparable.

Registered periodic in E2E_TIERS + touchfiles (with the auq-repetition-cut
A/B); periodic detach timeout raised to the new shard-census floor. Free
selftest (8 tests, zero API) pins fixtures, extraction, arm asymmetry, diff
capture, judge plumbing, and the retry bound.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
Garry Tan
2026-08-28 02:07:56 +00:00
co-authored by Claude Fable 5
parent 781f46d025
commit 4c20eca33b
21 changed files with 1076 additions and 10 deletions
+34
View File
@@ -131,6 +131,7 @@ export const E2E_TOUCHFILES: Record<string, string[]> = {
// numbered-option lists, multi-phase ordering, idempotency state echo).
'preamble-script-ab': ['bin/gstack-skill-start', 'bin/gstack-skill-end', 'scripts/resolvers/preamble/generate-preamble-bash.ts', 'scripts/resolvers/preamble/generate-brain-sync-block.ts', 'scripts/resolvers/preamble.ts', 'plan-ceo-review/**', 'test/helpers/auq-sdk-capture.ts', 'test/skill-e2e-preamble-script-ab.test.ts'],
'auq-format-gate': ['plan-ceo-review/**', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/preamble/generate-completeness-section.ts', 'scripts/resolvers/preamble.ts', 'test/helpers/auq-sdk-capture.ts', 'test/helpers/session-runner.ts', 'test/helpers/llm-judge.ts'],
'auq-repetition-cut-ab': ['scripts/resolvers/preamble/generate-ask-user-format.ts', 'plan-ceo-review/**', 'test/helpers/auq-sdk-capture.ts', 'test/skill-e2e-auq-repetition-cut-ab.test.ts'],
'plan-ceo-mode-routing': ['plan-ceo-review/**', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/preamble.ts', 'test/helpers/claude-pty-runner.ts', 'test/skill-e2e-plan-ceo-mode-routing.test.ts'],
'plan-design-with-ui-scope': ['plan-design-review/**', 'test/fixtures/plans/ui-heavy-feature.md', 'test/helpers/claude-pty-runner.ts', 'test/skill-e2e-plan-design-with-ui.test.ts'],
'budget-regression-pty': ['test/helpers/eval-store.ts', 'test/skill-budget-regression.test.ts'],
@@ -439,6 +440,32 @@ export const E2E_TOUCHFILES: Record<string, string[]> = {
'test/skill-e2e-gbrain-roundtrip-local.test.ts',
],
// WS2 arm benchmark — with-skill vs without-skill agentic arms scored on
// the git diff left behind (research instrument, never a release gate).
// Fires when the behavioral layer under test (reuse ladder + bounded
// closer resolvers), the judge, the fixtures, or the harness change.
'arm-benchmark-native-overbuild': [
'scripts/resolvers/preamble/generate-search-before-building.ts',
'scripts/resolvers/preamble/generate-voice-directive.ts',
'test/fixtures/arm-benchmark/**',
'test/helpers/llm-judge.ts',
'test/skill-e2e-arm-benchmark.test.ts',
],
'arm-benchmark-crud-endpoint': [
'scripts/resolvers/preamble/generate-search-before-building.ts',
'scripts/resolvers/preamble/generate-voice-directive.ts',
'test/fixtures/arm-benchmark/**',
'test/helpers/llm-judge.ts',
'test/skill-e2e-arm-benchmark.test.ts',
],
'arm-benchmark-bugfix-decoys': [
'scripts/resolvers/preamble/generate-search-before-building.ts',
'scripts/resolvers/preamble/generate-voice-directive.ts',
'test/fixtures/arm-benchmark/**',
'test/helpers/llm-judge.ts',
'test/skill-e2e-arm-benchmark.test.ts',
],
};
/**
@@ -546,6 +573,7 @@ export const E2E_TIERS: Record<string, 'gate' | 'periodic'> = {
// gate: cheap, deterministic, run on every PR
// periodic: long-running or expensive (>$3/run), run weekly
'preamble-script-ab': 'periodic', // Phase 1-3 A/B: script vs inline preamble; demoted post-Phase-3 (OV7)
'auq-repetition-cut-ab': 'periodic', // AUQ repetition-cut NOT-WORSE gate (passed pre-landing; re-runs on AUQ format changes)
'auq-format-gate': 'gate', // ~$0.50/run, SDK capture, single skill probe
'plan-ceo-mode-routing': 'periodic', // ~$3/run, deep navigation through 8-12 prior AskUserQuestions
'plan-design-with-ui-scope': 'gate', // ~$0.80/run
@@ -752,6 +780,12 @@ export const E2E_TIERS: Record<string, 'gate' | 'periodic'> = {
'ios-qa-device': 'periodic',
// /spec end-to-end PTY pipeline (paid, non-deterministic — periodic-tier).
'spec-execute': 'periodic',
// WS2 arm benchmark — periodic: full build-shaped agentic workflows, paid,
// non-deterministic by construction (research instrument, not a gate).
'arm-benchmark-native-overbuild': 'periodic',
'arm-benchmark-crud-endpoint': 'periodic',
'arm-benchmark-bugfix-decoys': 'periodic',
};
/**