mirror of
https://github.com/garrytan/gstack.git
synced 2026-09-14 08:59:01 +02:00
refactor(evals): arm-benchmark selftest runs FREE on every PR
The selftest lived inside the paid skill-e2e-* file, so fixture-integrity and plumbing pins executed weekly at best — a broken fixture would ship past every gating check and be discovered when the periodic run burned money on a dead instrument. Harness extracted to test/helpers/arm-benchmark-harness.ts, selftest to test/arm-benchmark-selftest.test.ts (free suite). Touchfiles: harness added to the three benchmark dep lists; the auq-repetition-cut-ab tier comment now states the MANUAL re-run obligation honestly (periodic runs force EVALS_ALL, so dep lists cannot auto-trigger it). Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Fable 5
parent
760554d0ac
commit
2ee61ab455
@@ -0,0 +1,208 @@
|
|||||||
|
/**
|
||||||
|
* Arm-benchmark selftest — FREE (no API key, no model, no spend), runs in
|
||||||
|
* `bun run test` on every PR. Pins fixture integrity (planted traps still
|
||||||
|
* open, decoy credentials obviously fake), skill extraction, arm-install
|
||||||
|
* asymmetry, diff-capture round trips, judge prompt construction/injection
|
||||||
|
* framing, parse plumbing, and the bounded retry — so the paid periodic
|
||||||
|
* benchmark never burns money on a broken instrument.
|
||||||
|
*/
|
||||||
|
import { describe, test, expect } from 'bun:test';
|
||||||
|
import {
|
||||||
|
TASKS, FIXTURES, SKILL_NAME,
|
||||||
|
buildBehavioralSkill, run, setupArm, parseDiffStat, captureStagedDiff,
|
||||||
|
} from './helpers/arm-benchmark-harness';
|
||||||
|
import {
|
||||||
|
armJudge, buildArmJudgePrompt, parseArmJudgeResponse,
|
||||||
|
ARM_JUDGE_ATTEMPTS, callJudge,
|
||||||
|
} from './helpers/llm-judge';
|
||||||
|
import * as fs from 'fs';
|
||||||
|
import * as path from 'path';
|
||||||
|
|
||||||
|
describe('arm benchmark selftest (free, no API)', () => {
|
||||||
|
test('fixtures exist with their planted content; decoy credentials are obviously fake', () => {
|
||||||
|
for (const task of TASKS) {
|
||||||
|
expect(fs.existsSync(path.join(FIXTURES, task.fixture))).toBe(true);
|
||||||
|
}
|
||||||
|
// Task 1: the form exists and has NO date input yet (the trap is open).
|
||||||
|
const html = fs.readFileSync(path.join(FIXTURES, 'native-overbuild', 'index.html'), 'utf-8');
|
||||||
|
expect(html).toContain('booking-form');
|
||||||
|
expect(html).not.toContain('type="date"');
|
||||||
|
// Task 2: GET/POST exist, DELETE does not.
|
||||||
|
const app = fs.readFileSync(path.join(FIXTURES, 'crud-endpoint', 'app.js'), 'utf-8');
|
||||||
|
expect(app).toContain("'GET'");
|
||||||
|
expect(app).toContain("'POST'");
|
||||||
|
expect(app).not.toContain('DELETE');
|
||||||
|
// Task 3: planted bug is live and the decoy credential can't trip a
|
||||||
|
// live-format scanner.
|
||||||
|
const price = fs.readFileSync(path.join(FIXTURES, 'bugfix-decoys', 'src', 'format-price.js'), 'utf-8');
|
||||||
|
expect(price).toContain("'$' + dollars + '.' + rem");
|
||||||
|
const config = fs.readFileSync(path.join(FIXTURES, 'bugfix-decoys', 'src', 'config.js'), 'utf-8');
|
||||||
|
expect(config).toContain('not-a-real-credential');
|
||||||
|
expect(config).not.toMatch(/sk-[a-zA-Z0-9]{16,}/);
|
||||||
|
// Decoy over-build invitations are planted.
|
||||||
|
const readme = fs.readFileSync(path.join(FIXTURES, 'bugfix-decoys', 'README.md'), 'utf-8');
|
||||||
|
expect(readme).toContain('plugin architecture');
|
||||||
|
});
|
||||||
|
|
||||||
|
test('behavioral skill is an extraction (ladder + bounded closer), not a whole-file copy', () => {
|
||||||
|
const skill = buildBehavioralSkill();
|
||||||
|
expect(skill).toContain(`name: ${SKILL_NAME}`);
|
||||||
|
expect(skill).toContain('## Search Before Building');
|
||||||
|
expect(skill).toContain('first rung that holds');
|
||||||
|
expect(skill).toContain('## Voice');
|
||||||
|
expect(skill).toContain('**Bounded closer.**');
|
||||||
|
// Telemetry tail stripped: a hermetic child must not write to the
|
||||||
|
// operator's real ~/.gstack.
|
||||||
|
expect(skill).not.toContain('Eureka');
|
||||||
|
// Extraction proof: none of ship's workflow rode along.
|
||||||
|
expect(skill).not.toContain('## Preamble (run first)');
|
||||||
|
expect(skill).not.toContain('Review Readiness');
|
||||||
|
expect(skill.length).toBeLessThan(8192);
|
||||||
|
});
|
||||||
|
|
||||||
|
test('with-arm installs the skill + routing line; without-arm installs neither; both get git + bare origin', () => {
|
||||||
|
const withArm = setupArm(TASKS[0], 'with-skill');
|
||||||
|
const withoutArm = setupArm(TASKS[0], 'without-skill');
|
||||||
|
try {
|
||||||
|
const skillPath = path.join(withArm.dir, '.claude', 'skills', SKILL_NAME, 'SKILL.md');
|
||||||
|
expect(fs.existsSync(skillPath)).toBe(true);
|
||||||
|
expect(fs.readFileSync(path.join(withArm.dir, 'CLAUDE.md'), 'utf-8')).toContain('## Skill routing');
|
||||||
|
|
||||||
|
expect(fs.existsSync(path.join(withoutArm.dir, '.claude'))).toBe(false);
|
||||||
|
expect(fs.readFileSync(path.join(withoutArm.dir, 'CLAUDE.md'), 'utf-8')).not.toContain('Skill routing');
|
||||||
|
|
||||||
|
// Both arms: seeded commit + working bare origin (merge-base-style
|
||||||
|
// commands must work inside the arm).
|
||||||
|
for (const arm of [withArm, withoutArm]) {
|
||||||
|
expect(run('git', ['rev-parse', 'HEAD'], arm.dir).trim()).toMatch(/^[0-9a-f]{40}$/);
|
||||||
|
expect(run('git', ['remote', 'get-url', 'origin'], arm.dir).trim()).toBe(arm.originDir);
|
||||||
|
expect(run('git', ['merge-base', 'origin/main', 'HEAD'], arm.dir).trim()).toMatch(/^[0-9a-f]{40}$/);
|
||||||
|
}
|
||||||
|
} finally {
|
||||||
|
for (const arm of [withArm, withoutArm]) {
|
||||||
|
fs.rmSync(arm.dir, { recursive: true, force: true });
|
||||||
|
fs.rmSync(arm.originDir, { recursive: true, force: true });
|
||||||
|
}
|
||||||
|
}
|
||||||
|
});
|
||||||
|
|
||||||
|
test('diff capture: stat parsing + a real zero-diff and non-zero-diff round trip', () => {
|
||||||
|
expect(parseDiffStat(' 3 files changed, 120 insertions(+), 4 deletions(-)\n'))
|
||||||
|
.toEqual({ filesChanged: 3, insertions: 120, deletions: 4, net: 116 });
|
||||||
|
expect(parseDiffStat(' 1 file changed, 2 insertions(+)\n'))
|
||||||
|
.toEqual({ filesChanged: 1, insertions: 2, deletions: 0, net: 2 });
|
||||||
|
expect(parseDiffStat(''))
|
||||||
|
.toEqual({ filesChanged: 0, insertions: 0, deletions: 0, net: 0 });
|
||||||
|
|
||||||
|
const arm = setupArm(TASKS[2], 'without-skill');
|
||||||
|
try {
|
||||||
|
// Zero-diff arm: a VALID cell, zeros across the board.
|
||||||
|
const clean = captureStagedDiff(arm.dir, arm.seedSha);
|
||||||
|
expect(clean.filesChanged).toBe(0);
|
||||||
|
expect(clean.net).toBe(0);
|
||||||
|
expect(clean.patch.trim()).toBe('');
|
||||||
|
|
||||||
|
// Modify + add a file: counts appear, patch carries the change.
|
||||||
|
fs.appendFileSync(path.join(arm.dir, 'README.md'), 'appended line\n');
|
||||||
|
fs.writeFileSync(path.join(arm.dir, 'new-file.txt'), 'one\ntwo\n');
|
||||||
|
const dirty = captureStagedDiff(arm.dir, arm.seedSha);
|
||||||
|
expect(dirty.filesChanged).toBe(2);
|
||||||
|
expect(dirty.insertions).toBe(3);
|
||||||
|
expect(dirty.deletions).toBe(0);
|
||||||
|
expect(dirty.net).toBe(3);
|
||||||
|
expect(dirty.patch).toContain('appended line');
|
||||||
|
} finally {
|
||||||
|
fs.rmSync(arm.dir, { recursive: true, force: true });
|
||||||
|
fs.rmSync(arm.originDir, { recursive: true, force: true });
|
||||||
|
}
|
||||||
|
});
|
||||||
|
|
||||||
|
test('judge prompt construction embeds the rubric, the ticket, and the reference diffs', () => {
|
||||||
|
const goodDiff = fs.readFileSync(path.join(FIXTURES, 'reference', 'good-diff.patch'), 'utf-8');
|
||||||
|
const badDiff = fs.readFileSync(path.join(FIXTURES, 'reference', 'bad-diff.patch'), 'utf-8');
|
||||||
|
for (const diff of [goodDiff, badDiff]) {
|
||||||
|
const prompt = buildArmJudgePrompt(TASKS[0].ticket, diff, 'pinned0000');
|
||||||
|
expect(prompt).toContain('<<<UNTRUSTED_DIFF_pinned0000>>>');
|
||||||
|
expect(prompt).toContain('<<<END_UNTRUSTED_DIFF_pinned0000>>>');
|
||||||
|
expect(prompt).toContain(diff);
|
||||||
|
expect(prompt).toContain(TASKS[0].ticket);
|
||||||
|
expect(prompt).toContain('0-3 scale');
|
||||||
|
expect(prompt).toContain('Coverage is NOT over-engineering');
|
||||||
|
expect(prompt).toContain('MUST name the specific class, function, file, or pattern');
|
||||||
|
expect(prompt).toContain('construct MUST be exactly "none"');
|
||||||
|
}
|
||||||
|
// The reference diffs are what the rubric anchors describe: the bad diff
|
||||||
|
// carries a hand-rolled widget replacing a native element, the good one
|
||||||
|
// uses the platform.
|
||||||
|
expect(badDiff).toContain('class CalendarWidget');
|
||||||
|
expect(goodDiff).toContain('type="date"');
|
||||||
|
|
||||||
|
// Injection hardening: without an explicit sentinel, each call gets its
|
||||||
|
// own random block markers — an arm diff cannot pre-write a closing
|
||||||
|
// marker it has never seen.
|
||||||
|
const a = buildArmJudgePrompt(TASKS[0].ticket, goodDiff);
|
||||||
|
const b = buildArmJudgePrompt(TASKS[0].ticket, goodDiff);
|
||||||
|
const marker = (p: string) => /<<<UNTRUSTED_DIFF_([a-z0-9]+)>>>/.exec(p)?.[1];
|
||||||
|
expect(marker(a)).toBeTruthy();
|
||||||
|
expect(marker(b)).toBeTruthy();
|
||||||
|
expect(marker(a)).not.toBe(marker(b));
|
||||||
|
});
|
||||||
|
|
||||||
|
test('judge response parsing: reference-shaped verdicts accepted, malformed rejected', () => {
|
||||||
|
// Canned verdicts the judge should return for the reference diffs.
|
||||||
|
const goodVerdict = parseArmJudgeResponse({
|
||||||
|
over_engineering: 0,
|
||||||
|
construct: 'none',
|
||||||
|
reasoning: 'Native date input with a min attribute; nothing unrequested.',
|
||||||
|
});
|
||||||
|
expect(goodVerdict.over_engineering).toBe(0);
|
||||||
|
expect(goodVerdict.construct).toBe('none');
|
||||||
|
|
||||||
|
const badVerdict = parseArmJudgeResponse({
|
||||||
|
over_engineering: 3,
|
||||||
|
construct: 'hand-rolled CalendarWidget + DatePickerFactory in calendar.js',
|
||||||
|
reasoning: 'A custom calendar widget layer replaces <input type="date">.',
|
||||||
|
});
|
||||||
|
expect(badVerdict.over_engineering).toBe(3);
|
||||||
|
expect(badVerdict.construct).toContain('CalendarWidget');
|
||||||
|
|
||||||
|
// Malformed shapes throw — that throw is what the bounded retry catches.
|
||||||
|
expect(() => parseArmJudgeResponse({ over_engineering: 7, construct: 'x' })).toThrow(/integer 0-3/);
|
||||||
|
expect(() => parseArmJudgeResponse({ over_engineering: 1.5, construct: 'x' })).toThrow(/integer 0-3/);
|
||||||
|
expect(() => parseArmJudgeResponse({ over_engineering: 2 })).toThrow(/construct missing/);
|
||||||
|
expect(() => parseArmJudgeResponse({ over_engineering: 2, construct: 'none' })).toThrow(/must name the specific construct/);
|
||||||
|
expect(() => parseArmJudgeResponse({ over_engineering: 0, construct: 'a helper' })).toThrow(/construct "none"/);
|
||||||
|
expect(() => parseArmJudgeResponse(null)).toThrow();
|
||||||
|
});
|
||||||
|
|
||||||
|
test('armJudge: zero diff scores deterministically as none with no API call', async () => {
|
||||||
|
// No ANTHROPIC client is ever constructed on this path — safe keyless.
|
||||||
|
const score = await armJudge(TASKS[0].ticket, ' \n');
|
||||||
|
expect(score.over_engineering).toBe(0);
|
||||||
|
expect(score.construct).toBe('none');
|
||||||
|
});
|
||||||
|
|
||||||
|
test('armJudge: bounded retry-on-malformed — recovers once, then gives up', async () => {
|
||||||
|
// Malformed first, valid second: recovers within the 2-attempt bound.
|
||||||
|
let calls = 0;
|
||||||
|
const flaky = (async () => {
|
||||||
|
calls++;
|
||||||
|
return calls === 1
|
||||||
|
? { over_engineering: 9, construct: 'garbage' }
|
||||||
|
: { over_engineering: 2, construct: 'repository layer in app.js', reasoning: 'ok' };
|
||||||
|
}) as unknown as typeof callJudge;
|
||||||
|
const recovered = await armJudge('ticket', 'diff --git a/x b/x\n+1\n', { call: flaky });
|
||||||
|
expect(recovered.over_engineering).toBe(2);
|
||||||
|
expect(calls).toBe(ARM_JUDGE_ATTEMPTS);
|
||||||
|
|
||||||
|
// Always malformed: throws after exactly ARM_JUDGE_ATTEMPTS attempts.
|
||||||
|
let badCalls = 0;
|
||||||
|
const alwaysBad = (async () => {
|
||||||
|
badCalls++;
|
||||||
|
return { nonsense: true };
|
||||||
|
}) as unknown as typeof callJudge;
|
||||||
|
await expect(armJudge('ticket', 'diff --git a/x b/x\n+1\n', { call: alwaysBad }))
|
||||||
|
.rejects.toThrow(/no well-formed verdict after 2 attempts/);
|
||||||
|
expect(badCalls).toBe(ARM_JUDGE_ATTEMPTS);
|
||||||
|
});
|
||||||
|
});
|
||||||
@@ -0,0 +1,219 @@
|
|||||||
|
/**
|
||||||
|
* Arm-benchmark harness — shared by the paid periodic benchmark
|
||||||
|
* (test/skill-e2e-arm-benchmark.test.ts) and the FREE selftest
|
||||||
|
* (test/arm-benchmark-selftest.test.ts). Extracted so the fixture-integrity
|
||||||
|
* and plumbing pins run in `bun run test` on every PR: the paid file matches
|
||||||
|
* the skill-e2e-* paid glob, so a selftest living inside it executed weekly
|
||||||
|
* at best — a broken fixture would ship past every gating check and be
|
||||||
|
* discovered only when the periodic run burned money on a dead instrument.
|
||||||
|
*/
|
||||||
|
import { ROOT, copyDirSync } from './e2e-helpers';
|
||||||
|
import { extractSkillSections } from './skill-fixture';
|
||||||
|
import { spawnSync } from 'child_process';
|
||||||
|
import * as fs from 'fs';
|
||||||
|
import * as path from 'path';
|
||||||
|
import * as os from 'os';
|
||||||
|
|
||||||
|
// --- Named per-arm constants (plan: defaults of 15 turns/120s are nowhere
|
||||||
|
// near enough for a build-shaped ticket: read fixture, implement, run tests).
|
||||||
|
export const ARM_MAX_TURNS = 40;
|
||||||
|
export const ARM_TIMEOUT_MS = 8 * 60_000;
|
||||||
|
/** Max diff bytes sent to the over-engineering judge. Truncation is loud
|
||||||
|
* (logged + suffixed onto judge_reasoning) — a clipped patch can hide the
|
||||||
|
* construct being scored, so a silent cap would corrupt cells invisibly. */
|
||||||
|
export const ARM_JUDGE_DIFF_CAP = 30_000;
|
||||||
|
// Skill tool in BOTH arms so the tool surface is symmetric — the without-arm
|
||||||
|
// simply has nothing installed to invoke. No Agent: build-discipline
|
||||||
|
// dispatches no subagents.
|
||||||
|
export const ARM_ALLOWED_TOOLS = ['Bash', 'Read', 'Write', 'Edit', 'Glob', 'Grep', 'Skill'];
|
||||||
|
// Two concurrent arms + up to two judge calls + fixture setup.
|
||||||
|
export const TASK_TEST_TIMEOUT_MS = ARM_TIMEOUT_MS + 240_000;
|
||||||
|
|
||||||
|
export const SKILL_NAME = 'build-discipline';
|
||||||
|
export const FIXTURES = path.join(ROOT, 'test', 'fixtures', 'arm-benchmark');
|
||||||
|
|
||||||
|
export type Arm = 'with-skill' | 'without-skill';
|
||||||
|
|
||||||
|
export interface ArmTask {
|
||||||
|
/** E2E_TOUCHFILES / E2E_TIERS key. Quoted literals below keep the parent
|
||||||
|
* shard mapper (test-paid-shards) able to attribute this file. */
|
||||||
|
key: string;
|
||||||
|
fixture: string;
|
||||||
|
ticket: string;
|
||||||
|
}
|
||||||
|
|
||||||
|
// The 3 named tasks (plan: one native-platform over-build trap, one CRUD
|
||||||
|
// endpoint, one bug-fix with planted decoy over-build invitations).
|
||||||
|
export const TASKS: ArmTask[] = [
|
||||||
|
{
|
||||||
|
key: 'arm-benchmark-native-overbuild',
|
||||||
|
fixture: 'native-overbuild',
|
||||||
|
ticket: `Ticket: hikers need to pick their hike date when booking.
|
||||||
|
|
||||||
|
You are in a small git repo containing a static site (index.html, app.js, styles.css — no build step).
|
||||||
|
Add a date field to the booking form: it must be required, must not allow choosing a past date, and the confirmation message must include the chosen date.
|
||||||
|
|
||||||
|
Leave your changes uncommitted in the working tree.`,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
key: 'arm-benchmark-crud-endpoint',
|
||||||
|
fixture: 'crud-endpoint',
|
||||||
|
ticket: `Ticket: users need to delete notes.
|
||||||
|
|
||||||
|
You are in a small git repo containing an in-memory notes API (app.js, wired to HTTP in server.js).
|
||||||
|
Add DELETE /notes/:id: respond 204 on success and 404 for an unknown id, and cover the new endpoint in run-tests.js. Verify with: node run-tests.js
|
||||||
|
|
||||||
|
Leave your changes uncommitted in the working tree.`,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
key: 'arm-benchmark-bugfix-decoys',
|
||||||
|
fixture: 'bugfix-decoys',
|
||||||
|
ticket: `Bug report: receipts print $10.5 for a $10.05 item.
|
||||||
|
|
||||||
|
You are in a small git repo. \`node run-tests.js\` currently fails on formatPrice(1005).
|
||||||
|
Fix the bug so all tests pass. Verify with: node run-tests.js
|
||||||
|
|
||||||
|
Leave your changes uncommitted in the working tree.`,
|
||||||
|
},
|
||||||
|
];
|
||||||
|
|
||||||
|
// --- Skill under test: extracted behavioral layer ---
|
||||||
|
|
||||||
|
/** Drop the Eureka telemetry tail from the extracted Search Before Building
|
||||||
|
* section: it appends to the OPERATOR's real ~/.gstack from inside a
|
||||||
|
* hermetic child, and telemetry is not the behavior under test. */
|
||||||
|
export function stripEureka(text: string): string {
|
||||||
|
const start = text.indexOf('**Eureka:**');
|
||||||
|
if (start === -1) return text;
|
||||||
|
const next = text.indexOf('\n## ', start);
|
||||||
|
return text.slice(0, start) + (next === -1 ? '' : text.slice(next + 1));
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Assemble the behavioral-layer skill: the WS3 reuse ladder (## Search Before
|
||||||
|
* Building) + the WS7 bounded closer (## Voice), extracted from the rendered
|
||||||
|
* ship/SKILL.md (tier 4 — carries both sections) and wrapped in this
|
||||||
|
* benchmark's own frontmatter. Extract, don't copy (CLAUDE.md rule).
|
||||||
|
*/
|
||||||
|
export function buildBehavioralSkill(): string {
|
||||||
|
const extracted = extractSkillSections(path.join(ROOT, 'ship'), ['Search Before Building', 'Voice']);
|
||||||
|
const body = stripEureka(extracted.replace(/^---\n[\s\S]*?\n---\n/, '')).trim();
|
||||||
|
return `---
|
||||||
|
name: ${SKILL_NAME}
|
||||||
|
description: Build discipline for implementation tickets — the reuse ladder (stop at the first rung that holds) plus bounded completion reports. Invoke before implementing any ticket.
|
||||||
|
---
|
||||||
|
|
||||||
|
# Build discipline
|
||||||
|
|
||||||
|
Apply these rules to the implementation work you are about to do.
|
||||||
|
|
||||||
|
${body}
|
||||||
|
`;
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- Arm setup: fixture copy + optional skill install + git init + bare origin ---
|
||||||
|
|
||||||
|
export interface ArmDirs {
|
||||||
|
dir: string;
|
||||||
|
originDir: string;
|
||||||
|
/** The seed commit — the immutable diff base for harvest (an agent that
|
||||||
|
* disobeys "leave uncommitted" by committing AND pushing can move
|
||||||
|
* origin/main, but it cannot move a recorded SHA). */
|
||||||
|
seedSha: string;
|
||||||
|
}
|
||||||
|
|
||||||
|
export function run(cmd: string, args: string[], cwd: string): string {
|
||||||
|
// 64MB maxBuffer: the patch capture pipes the FULL staged diff through
|
||||||
|
// here, and the most over-built arm outcome (vendored dependency) is
|
||||||
|
// exactly the one the benchmark must not die on.
|
||||||
|
const r = spawnSync(cmd, args, { cwd, stdio: 'pipe', encoding: 'utf-8', timeout: 15_000, maxBuffer: 64 * 1024 * 1024 });
|
||||||
|
if (r.status !== 0) {
|
||||||
|
throw new Error(`${cmd} ${args.join(' ')} failed in ${cwd}: ${r.stderr || r.stdout}`);
|
||||||
|
}
|
||||||
|
return r.stdout ?? '';
|
||||||
|
}
|
||||||
|
|
||||||
|
export function setupArm(task: ArmTask, arm: Arm): ArmDirs {
|
||||||
|
const dir = fs.mkdtempSync(path.join(os.tmpdir(), `arm-${task.fixture}-${arm}-`));
|
||||||
|
copyDirSync(path.join(FIXTURES, task.fixture), dir);
|
||||||
|
|
||||||
|
const baseClaudeMd = '# Project\n\nSmall fixture repo for an implementation ticket. Run its checks with the command named in the ticket.\n';
|
||||||
|
if (arm === 'with-skill') {
|
||||||
|
const skillDir = path.join(dir, '.claude', 'skills', SKILL_NAME);
|
||||||
|
fs.mkdirSync(skillDir, { recursive: true });
|
||||||
|
fs.writeFileSync(path.join(skillDir, 'SKILL.md'), buildBehavioralSkill());
|
||||||
|
fs.writeFileSync(
|
||||||
|
path.join(dir, 'CLAUDE.md'),
|
||||||
|
baseClaudeMd
|
||||||
|
+ `\n## Skill routing\n\nBefore implementing any ticket, invoke the ${SKILL_NAME} skill via the Skill tool and follow it while you work.\n`,
|
||||||
|
);
|
||||||
|
} else {
|
||||||
|
fs.writeFileSync(path.join(dir, 'CLAUDE.md'), baseClaudeMd);
|
||||||
|
}
|
||||||
|
|
||||||
|
// node_modules never enters the harvest: an arm that npm-installs a
|
||||||
|
// dependency is a scoreable outcome, not a reason to stage 10k files.
|
||||||
|
if (!fs.existsSync(path.join(dir, '.gitignore'))) {
|
||||||
|
fs.writeFileSync(path.join(dir, '.gitignore'), 'node_modules/\n');
|
||||||
|
}
|
||||||
|
|
||||||
|
run('git', ['init', '-b', 'main'], dir);
|
||||||
|
run('git', ['config', 'user.email', 'arm-bench@example.com'], dir);
|
||||||
|
run('git', ['config', 'user.name', 'Arm Bench'], dir);
|
||||||
|
run('git', ['config', 'commit.gpgsign', 'false'], dir);
|
||||||
|
run('git', ['add', '-A'], dir);
|
||||||
|
run('git', ['commit', '-m', 'seed fixture'], dir);
|
||||||
|
const seedSha = run('git', ['rev-parse', 'HEAD'], dir).trim();
|
||||||
|
|
||||||
|
// Local bare origin so merge-base-style commands work inside the arm.
|
||||||
|
const originDir = fs.mkdtempSync(path.join(os.tmpdir(), `arm-${task.fixture}-${arm}-origin-`));
|
||||||
|
run('git', ['init', '--bare', '-b', 'main'], originDir);
|
||||||
|
run('git', ['remote', 'add', 'origin', originDir], dir);
|
||||||
|
run('git', ['push', '-u', 'origin', 'main'], dir);
|
||||||
|
|
||||||
|
return { dir, originDir, seedSha };
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- Diff capture: git add -A && git diff --cached --stat (plan spec) ---
|
||||||
|
|
||||||
|
export interface DiffHarvest {
|
||||||
|
filesChanged: number;
|
||||||
|
insertions: number;
|
||||||
|
deletions: number;
|
||||||
|
net: number;
|
||||||
|
stat: string;
|
||||||
|
patch: string;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Parse the summary line of `git diff --stat`. Empty stat = zero-diff
|
||||||
|
* (a VALID cell, not an error). */
|
||||||
|
export function parseDiffStat(stat: string): Pick<DiffHarvest, 'filesChanged' | 'insertions' | 'deletions' | 'net'> {
|
||||||
|
const line = stat.trim().split('\n').pop() ?? '';
|
||||||
|
const files = line.match(/(\d+) files? changed/);
|
||||||
|
const ins = line.match(/(\d+) insertions?\(\+\)/);
|
||||||
|
const del = line.match(/(\d+) deletions?\(-\)/);
|
||||||
|
const insertions = ins ? Number(ins[1]) : 0;
|
||||||
|
const deletions = del ? Number(del[1]) : 0;
|
||||||
|
return {
|
||||||
|
filesChanged: files ? Number(files[1]) : 0,
|
||||||
|
insertions,
|
||||||
|
deletions,
|
||||||
|
net: insertions - deletions,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Rung 2: three lines of git beat a generalized manager (WorktreeManager only
|
||||||
|
* harvests worktrees it created from the gstack repo — it cannot harvest
|
||||||
|
* synthetic fixtures). Diffing the index against the RECORDED seed SHA (not
|
||||||
|
* origin/main, which an agent that commits AND pushes can move; not HEAD,
|
||||||
|
* which a plain commit moves) keeps the capture honest under every flavor of
|
||||||
|
* "leave uncommitted" disobedience.
|
||||||
|
*/
|
||||||
|
export function captureStagedDiff(dir: string, seedSha: string): DiffHarvest {
|
||||||
|
run('git', ['add', '-A'], dir);
|
||||||
|
const stat = run('git', ['diff', '--cached', seedSha, '--stat'], dir);
|
||||||
|
const patch = run('git', ['diff', '--cached', seedSha], dir);
|
||||||
|
return { ...parseDiffStat(stat), stat: stat.trim(), patch };
|
||||||
|
}
|
||||||
|
|
||||||
@@ -454,6 +454,7 @@ export const E2E_TOUCHFILES: Record<string, string[]> = {
|
|||||||
'scripts/resolvers/preamble/generate-voice-directive.ts',
|
'scripts/resolvers/preamble/generate-voice-directive.ts',
|
||||||
'test/fixtures/arm-benchmark/**',
|
'test/fixtures/arm-benchmark/**',
|
||||||
'test/helpers/llm-judge.ts',
|
'test/helpers/llm-judge.ts',
|
||||||
|
'test/helpers/arm-benchmark-harness.ts',
|
||||||
'test/skill-e2e-arm-benchmark.test.ts',
|
'test/skill-e2e-arm-benchmark.test.ts',
|
||||||
'ship/SKILL.md',
|
'ship/SKILL.md',
|
||||||
],
|
],
|
||||||
@@ -462,6 +463,7 @@ export const E2E_TOUCHFILES: Record<string, string[]> = {
|
|||||||
'scripts/resolvers/preamble/generate-voice-directive.ts',
|
'scripts/resolvers/preamble/generate-voice-directive.ts',
|
||||||
'test/fixtures/arm-benchmark/**',
|
'test/fixtures/arm-benchmark/**',
|
||||||
'test/helpers/llm-judge.ts',
|
'test/helpers/llm-judge.ts',
|
||||||
|
'test/helpers/arm-benchmark-harness.ts',
|
||||||
'test/skill-e2e-arm-benchmark.test.ts',
|
'test/skill-e2e-arm-benchmark.test.ts',
|
||||||
'ship/SKILL.md',
|
'ship/SKILL.md',
|
||||||
],
|
],
|
||||||
@@ -470,6 +472,7 @@ export const E2E_TOUCHFILES: Record<string, string[]> = {
|
|||||||
'scripts/resolvers/preamble/generate-voice-directive.ts',
|
'scripts/resolvers/preamble/generate-voice-directive.ts',
|
||||||
'test/fixtures/arm-benchmark/**',
|
'test/fixtures/arm-benchmark/**',
|
||||||
'test/helpers/llm-judge.ts',
|
'test/helpers/llm-judge.ts',
|
||||||
|
'test/helpers/arm-benchmark-harness.ts',
|
||||||
'test/skill-e2e-arm-benchmark.test.ts',
|
'test/skill-e2e-arm-benchmark.test.ts',
|
||||||
'ship/SKILL.md',
|
'ship/SKILL.md',
|
||||||
],
|
],
|
||||||
@@ -581,7 +584,7 @@ export const E2E_TIERS: Record<string, 'gate' | 'periodic'> = {
|
|||||||
// gate: cheap, deterministic, run on every PR
|
// gate: cheap, deterministic, run on every PR
|
||||||
// periodic: long-running or expensive (>$3/run), run weekly
|
// periodic: long-running or expensive (>$3/run), run weekly
|
||||||
'preamble-script-ab': 'periodic', // Phase 1-3 A/B: script vs inline preamble; demoted post-Phase-3 (OV7)
|
'preamble-script-ab': 'periodic', // Phase 1-3 A/B: script vs inline preamble; demoted post-Phase-3 (OV7)
|
||||||
'auq-repetition-cut-ab': 'periodic', // AUQ repetition-cut NOT-WORSE gate (passed pre-landing; re-runs on AUQ format changes)
|
'auq-repetition-cut-ab': 'periodic', // AUQ repetition-cut NOT-WORSE gate (passed pre-landing). Periodic runs force EVALS_ALL, so the dep list cannot auto-trigger it — an AUQ format edit carries a MANUAL re-run obligation (bun test test/skill-e2e-auq-repetition-cut-ab.test.ts with EVALS=1 EVALS_TIER=periodic)
|
||||||
'auq-format-gate': 'gate', // ~$0.50/run, SDK capture, single skill probe
|
'auq-format-gate': 'gate', // ~$0.50/run, SDK capture, single skill probe
|
||||||
'plan-ceo-mode-routing': 'periodic', // ~$3/run, deep navigation through 8-12 prior AskUserQuestions
|
'plan-ceo-mode-routing': 'periodic', // ~$3/run, deep navigation through 8-12 prior AskUserQuestions
|
||||||
'plan-design-with-ui-scope': 'gate', // ~$0.80/run
|
'plan-design-with-ui-scope': 'gate', // ~$0.80/run
|
||||||
|
|||||||
@@ -25,233 +25,31 @@
|
|||||||
* cell: excluded from aggregates, surfaced in the run report, never
|
* cell: excluded from aggregates, surfaced in the run report, never
|
||||||
* silently dropped.
|
* silently dropped.
|
||||||
*
|
*
|
||||||
* The selftest describe at the bottom is FREE (no API): fixture integrity,
|
* The harness (tasks, fixtures, skill assembly, arm setup, diff capture)
|
||||||
* skill extraction, arm installation asymmetry, diff-capture plumbing, and
|
* lives in test/helpers/arm-benchmark-harness.ts, shared with the FREE
|
||||||
* the judge's prompt-construction/parse path on reference good/bad diffs.
|
* selftest at test/arm-benchmark-selftest.test.ts — which runs in
|
||||||
* Everything needing a live model sits inside the EVALS_TIER=periodic
|
* `bun run test` on every PR so this paid instrument can never burn money on
|
||||||
* describes above it.
|
* broken fixtures or plumbing. Everything needing a live model is here,
|
||||||
|
* inside the EVALS_TIER=periodic describes.
|
||||||
*/
|
*/
|
||||||
|
|
||||||
import { describe, test, expect, afterAll } from 'bun:test';
|
import { describe, test, expect, afterAll } from 'bun:test';
|
||||||
import { runSkillTest } from './helpers/session-runner';
|
import { runSkillTest } from './helpers/session-runner';
|
||||||
import type { SkillTestResult } from './helpers/session-runner';
|
import type { SkillTestResult } from './helpers/session-runner';
|
||||||
import {
|
import {
|
||||||
ROOT, runId, selectedTests, logCost, recordE2E,
|
runId, selectedTests, logCost, recordE2E,
|
||||||
createEvalCollector, finalizeEvalCollector, copyDirSync,
|
createEvalCollector, finalizeEvalCollector,
|
||||||
} from './helpers/e2e-helpers';
|
} from './helpers/e2e-helpers';
|
||||||
import { describeE2ETier } from './helpers/e2e-gate';
|
import { describeE2ETier } from './helpers/e2e-gate';
|
||||||
import { extractSkillSections } from './helpers/skill-fixture';
|
import { armJudge, type ArmJudgeScore } from './helpers/llm-judge';
|
||||||
import {
|
import {
|
||||||
armJudge, buildArmJudgePrompt, parseArmJudgeResponse,
|
ARM_MAX_TURNS, ARM_TIMEOUT_MS, ARM_JUDGE_DIFF_CAP, ARM_ALLOWED_TOOLS,
|
||||||
ARM_JUDGE_ATTEMPTS, callJudge, type ArmJudgeScore,
|
TASK_TEST_TIMEOUT_MS, SKILL_NAME, TASKS,
|
||||||
} from './helpers/llm-judge';
|
setupArm, captureStagedDiff,
|
||||||
import { spawnSync } from 'child_process';
|
type Arm, type ArmTask, type DiffHarvest,
|
||||||
|
} from './helpers/arm-benchmark-harness';
|
||||||
import * as fs from 'fs';
|
import * as fs from 'fs';
|
||||||
import * as path from 'path';
|
|
||||||
import * as os from 'os';
|
|
||||||
|
|
||||||
// --- Named per-arm constants (plan: defaults of 15 turns/120s are nowhere
|
|
||||||
// near enough for a build-shaped ticket: read fixture, implement, run tests).
|
|
||||||
const ARM_MAX_TURNS = 40;
|
|
||||||
const ARM_TIMEOUT_MS = 8 * 60_000;
|
|
||||||
/** Max diff bytes sent to the over-engineering judge. Truncation is loud
|
|
||||||
* (logged + suffixed onto judge_reasoning) — a clipped patch can hide the
|
|
||||||
* construct being scored, so a silent cap would corrupt cells invisibly. */
|
|
||||||
const ARM_JUDGE_DIFF_CAP = 30_000;
|
|
||||||
// Skill tool in BOTH arms so the tool surface is symmetric — the without-arm
|
|
||||||
// simply has nothing installed to invoke. No Agent: build-discipline
|
|
||||||
// dispatches no subagents.
|
|
||||||
const ARM_ALLOWED_TOOLS = ['Bash', 'Read', 'Write', 'Edit', 'Glob', 'Grep', 'Skill'];
|
|
||||||
// Two concurrent arms + up to two judge calls + fixture setup.
|
|
||||||
const TASK_TEST_TIMEOUT_MS = ARM_TIMEOUT_MS + 240_000;
|
|
||||||
|
|
||||||
const SKILL_NAME = 'build-discipline';
|
|
||||||
const FIXTURES = path.join(ROOT, 'test', 'fixtures', 'arm-benchmark');
|
|
||||||
|
|
||||||
type Arm = 'with-skill' | 'without-skill';
|
|
||||||
|
|
||||||
interface ArmTask {
|
|
||||||
/** E2E_TOUCHFILES / E2E_TIERS key. Quoted literals below keep the parent
|
|
||||||
* shard mapper (test-paid-shards) able to attribute this file. */
|
|
||||||
key: string;
|
|
||||||
fixture: string;
|
|
||||||
ticket: string;
|
|
||||||
}
|
|
||||||
|
|
||||||
// The 3 named tasks (plan: one native-platform over-build trap, one CRUD
|
|
||||||
// endpoint, one bug-fix with planted decoy over-build invitations).
|
|
||||||
const TASKS: ArmTask[] = [
|
|
||||||
{
|
|
||||||
key: 'arm-benchmark-native-overbuild',
|
|
||||||
fixture: 'native-overbuild',
|
|
||||||
ticket: `Ticket: hikers need to pick their hike date when booking.
|
|
||||||
|
|
||||||
You are in a small git repo containing a static site (index.html, app.js, styles.css — no build step).
|
|
||||||
Add a date field to the booking form: it must be required, must not allow choosing a past date, and the confirmation message must include the chosen date.
|
|
||||||
|
|
||||||
Leave your changes uncommitted in the working tree.`,
|
|
||||||
},
|
|
||||||
{
|
|
||||||
key: 'arm-benchmark-crud-endpoint',
|
|
||||||
fixture: 'crud-endpoint',
|
|
||||||
ticket: `Ticket: users need to delete notes.
|
|
||||||
|
|
||||||
You are in a small git repo containing an in-memory notes API (app.js, wired to HTTP in server.js).
|
|
||||||
Add DELETE /notes/:id: respond 204 on success and 404 for an unknown id, and cover the new endpoint in run-tests.js. Verify with: node run-tests.js
|
|
||||||
|
|
||||||
Leave your changes uncommitted in the working tree.`,
|
|
||||||
},
|
|
||||||
{
|
|
||||||
key: 'arm-benchmark-bugfix-decoys',
|
|
||||||
fixture: 'bugfix-decoys',
|
|
||||||
ticket: `Bug report: receipts print $10.5 for a $10.05 item.
|
|
||||||
|
|
||||||
You are in a small git repo. \`node run-tests.js\` currently fails on formatPrice(1005).
|
|
||||||
Fix the bug so all tests pass. Verify with: node run-tests.js
|
|
||||||
|
|
||||||
Leave your changes uncommitted in the working tree.`,
|
|
||||||
},
|
|
||||||
];
|
|
||||||
|
|
||||||
// --- Skill under test: extracted behavioral layer ---
|
|
||||||
|
|
||||||
/** Drop the Eureka telemetry tail from the extracted Search Before Building
|
|
||||||
* section: it appends to the OPERATOR's real ~/.gstack from inside a
|
|
||||||
* hermetic child, and telemetry is not the behavior under test. */
|
|
||||||
function stripEureka(text: string): string {
|
|
||||||
const start = text.indexOf('**Eureka:**');
|
|
||||||
if (start === -1) return text;
|
|
||||||
const next = text.indexOf('\n## ', start);
|
|
||||||
return text.slice(0, start) + (next === -1 ? '' : text.slice(next + 1));
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Assemble the behavioral-layer skill: the WS3 reuse ladder (## Search Before
|
|
||||||
* Building) + the WS7 bounded closer (## Voice), extracted from the rendered
|
|
||||||
* ship/SKILL.md (tier 4 — carries both sections) and wrapped in this
|
|
||||||
* benchmark's own frontmatter. Extract, don't copy (CLAUDE.md rule).
|
|
||||||
*/
|
|
||||||
function buildBehavioralSkill(): string {
|
|
||||||
const extracted = extractSkillSections(path.join(ROOT, 'ship'), ['Search Before Building', 'Voice']);
|
|
||||||
const body = stripEureka(extracted.replace(/^---\n[\s\S]*?\n---\n/, '')).trim();
|
|
||||||
return `---
|
|
||||||
name: ${SKILL_NAME}
|
|
||||||
description: Build discipline for implementation tickets — the reuse ladder (stop at the first rung that holds) plus bounded completion reports. Invoke before implementing any ticket.
|
|
||||||
---
|
|
||||||
|
|
||||||
# Build discipline
|
|
||||||
|
|
||||||
Apply these rules to the implementation work you are about to do.
|
|
||||||
|
|
||||||
${body}
|
|
||||||
`;
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- Arm setup: fixture copy + optional skill install + git init + bare origin ---
|
|
||||||
|
|
||||||
interface ArmDirs {
|
|
||||||
dir: string;
|
|
||||||
originDir: string;
|
|
||||||
/** The seed commit — the immutable diff base for harvest (an agent that
|
|
||||||
* disobeys "leave uncommitted" by committing AND pushing can move
|
|
||||||
* origin/main, but it cannot move a recorded SHA). */
|
|
||||||
seedSha: string;
|
|
||||||
}
|
|
||||||
|
|
||||||
function run(cmd: string, args: string[], cwd: string): string {
|
|
||||||
// 64MB maxBuffer: the patch capture pipes the FULL staged diff through
|
|
||||||
// here, and the most over-built arm outcome (vendored dependency) is
|
|
||||||
// exactly the one the benchmark must not die on.
|
|
||||||
const r = spawnSync(cmd, args, { cwd, stdio: 'pipe', encoding: 'utf-8', timeout: 15_000, maxBuffer: 64 * 1024 * 1024 });
|
|
||||||
if (r.status !== 0) {
|
|
||||||
throw new Error(`${cmd} ${args.join(' ')} failed in ${cwd}: ${r.stderr || r.stdout}`);
|
|
||||||
}
|
|
||||||
return r.stdout ?? '';
|
|
||||||
}
|
|
||||||
|
|
||||||
function setupArm(task: ArmTask, arm: Arm): ArmDirs {
|
|
||||||
const dir = fs.mkdtempSync(path.join(os.tmpdir(), `arm-${task.fixture}-${arm}-`));
|
|
||||||
copyDirSync(path.join(FIXTURES, task.fixture), dir);
|
|
||||||
|
|
||||||
const baseClaudeMd = '# Project\n\nSmall fixture repo for an implementation ticket. Run its checks with the command named in the ticket.\n';
|
|
||||||
if (arm === 'with-skill') {
|
|
||||||
const skillDir = path.join(dir, '.claude', 'skills', SKILL_NAME);
|
|
||||||
fs.mkdirSync(skillDir, { recursive: true });
|
|
||||||
fs.writeFileSync(path.join(skillDir, 'SKILL.md'), buildBehavioralSkill());
|
|
||||||
fs.writeFileSync(
|
|
||||||
path.join(dir, 'CLAUDE.md'),
|
|
||||||
baseClaudeMd
|
|
||||||
+ `\n## Skill routing\n\nBefore implementing any ticket, invoke the ${SKILL_NAME} skill via the Skill tool and follow it while you work.\n`,
|
|
||||||
);
|
|
||||||
} else {
|
|
||||||
fs.writeFileSync(path.join(dir, 'CLAUDE.md'), baseClaudeMd);
|
|
||||||
}
|
|
||||||
|
|
||||||
// node_modules never enters the harvest: an arm that npm-installs a
|
|
||||||
// dependency is a scoreable outcome, not a reason to stage 10k files.
|
|
||||||
if (!fs.existsSync(path.join(dir, '.gitignore'))) {
|
|
||||||
fs.writeFileSync(path.join(dir, '.gitignore'), 'node_modules/\n');
|
|
||||||
}
|
|
||||||
|
|
||||||
run('git', ['init', '-b', 'main'], dir);
|
|
||||||
run('git', ['config', 'user.email', 'arm-bench@example.com'], dir);
|
|
||||||
run('git', ['config', 'user.name', 'Arm Bench'], dir);
|
|
||||||
run('git', ['config', 'commit.gpgsign', 'false'], dir);
|
|
||||||
run('git', ['add', '-A'], dir);
|
|
||||||
run('git', ['commit', '-m', 'seed fixture'], dir);
|
|
||||||
const seedSha = run('git', ['rev-parse', 'HEAD'], dir).trim();
|
|
||||||
|
|
||||||
// Local bare origin so merge-base-style commands work inside the arm.
|
|
||||||
const originDir = fs.mkdtempSync(path.join(os.tmpdir(), `arm-${task.fixture}-${arm}-origin-`));
|
|
||||||
run('git', ['init', '--bare', '-b', 'main'], originDir);
|
|
||||||
run('git', ['remote', 'add', 'origin', originDir], dir);
|
|
||||||
run('git', ['push', '-u', 'origin', 'main'], dir);
|
|
||||||
|
|
||||||
return { dir, originDir, seedSha };
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- Diff capture: git add -A && git diff --cached --stat (plan spec) ---
|
|
||||||
|
|
||||||
interface DiffHarvest {
|
|
||||||
filesChanged: number;
|
|
||||||
insertions: number;
|
|
||||||
deletions: number;
|
|
||||||
net: number;
|
|
||||||
stat: string;
|
|
||||||
patch: string;
|
|
||||||
}
|
|
||||||
|
|
||||||
/** Parse the summary line of `git diff --stat`. Empty stat = zero-diff
|
|
||||||
* (a VALID cell, not an error). */
|
|
||||||
function parseDiffStat(stat: string): Pick<DiffHarvest, 'filesChanged' | 'insertions' | 'deletions' | 'net'> {
|
|
||||||
const line = stat.trim().split('\n').pop() ?? '';
|
|
||||||
const files = line.match(/(\d+) files? changed/);
|
|
||||||
const ins = line.match(/(\d+) insertions?\(\+\)/);
|
|
||||||
const del = line.match(/(\d+) deletions?\(-\)/);
|
|
||||||
const insertions = ins ? Number(ins[1]) : 0;
|
|
||||||
const deletions = del ? Number(del[1]) : 0;
|
|
||||||
return {
|
|
||||||
filesChanged: files ? Number(files[1]) : 0,
|
|
||||||
insertions,
|
|
||||||
deletions,
|
|
||||||
net: insertions - deletions,
|
|
||||||
};
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Rung 2: three lines of git beat a generalized manager (WorktreeManager only
|
|
||||||
* harvests worktrees it created from the gstack repo — it cannot harvest
|
|
||||||
* synthetic fixtures). Diffing the index against the RECORDED seed SHA (not
|
|
||||||
* origin/main, which an agent that commits AND pushes can move; not HEAD,
|
|
||||||
* which a plain commit moves) keeps the capture honest under every flavor of
|
|
||||||
* "leave uncommitted" disobedience.
|
|
||||||
*/
|
|
||||||
function captureStagedDiff(dir: string, seedSha: string): DiffHarvest {
|
|
||||||
run('git', ['add', '-A'], dir);
|
|
||||||
const stat = run('git', ['diff', '--cached', seedSha, '--stat'], dir);
|
|
||||||
const patch = run('git', ['diff', '--cached', seedSha], dir);
|
|
||||||
return { ...parseDiffStat(stat), stat: stat.trim(), patch };
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- Cell runner + reporting ---
|
// --- Cell runner + reporting ---
|
||||||
|
|
||||||
@@ -433,193 +231,3 @@ afterAll(async () => {
|
|||||||
await finalizeEvalCollector(evalCollector);
|
await finalizeEvalCollector(evalCollector);
|
||||||
});
|
});
|
||||||
|
|
||||||
// --- Selftest (FREE — no API key, no model, no spend) ---
|
|
||||||
|
|
||||||
describe('arm benchmark selftest (free, no API)', () => {
|
|
||||||
test('fixtures exist with their planted content; decoy credentials are obviously fake', () => {
|
|
||||||
for (const task of TASKS) {
|
|
||||||
expect(fs.existsSync(path.join(FIXTURES, task.fixture))).toBe(true);
|
|
||||||
}
|
|
||||||
// Task 1: the form exists and has NO date input yet (the trap is open).
|
|
||||||
const html = fs.readFileSync(path.join(FIXTURES, 'native-overbuild', 'index.html'), 'utf-8');
|
|
||||||
expect(html).toContain('booking-form');
|
|
||||||
expect(html).not.toContain('type="date"');
|
|
||||||
// Task 2: GET/POST exist, DELETE does not.
|
|
||||||
const app = fs.readFileSync(path.join(FIXTURES, 'crud-endpoint', 'app.js'), 'utf-8');
|
|
||||||
expect(app).toContain("'GET'");
|
|
||||||
expect(app).toContain("'POST'");
|
|
||||||
expect(app).not.toContain('DELETE');
|
|
||||||
// Task 3: planted bug is live and the decoy credential can't trip a
|
|
||||||
// live-format scanner.
|
|
||||||
const price = fs.readFileSync(path.join(FIXTURES, 'bugfix-decoys', 'src', 'format-price.js'), 'utf-8');
|
|
||||||
expect(price).toContain("'$' + dollars + '.' + rem");
|
|
||||||
const config = fs.readFileSync(path.join(FIXTURES, 'bugfix-decoys', 'src', 'config.js'), 'utf-8');
|
|
||||||
expect(config).toContain('not-a-real-credential');
|
|
||||||
expect(config).not.toMatch(/sk-[a-zA-Z0-9]{16,}/);
|
|
||||||
// Decoy over-build invitations are planted.
|
|
||||||
const readme = fs.readFileSync(path.join(FIXTURES, 'bugfix-decoys', 'README.md'), 'utf-8');
|
|
||||||
expect(readme).toContain('plugin architecture');
|
|
||||||
});
|
|
||||||
|
|
||||||
test('behavioral skill is an extraction (ladder + bounded closer), not a whole-file copy', () => {
|
|
||||||
const skill = buildBehavioralSkill();
|
|
||||||
expect(skill).toContain(`name: ${SKILL_NAME}`);
|
|
||||||
expect(skill).toContain('## Search Before Building');
|
|
||||||
expect(skill).toContain('first rung that holds');
|
|
||||||
expect(skill).toContain('## Voice');
|
|
||||||
expect(skill).toContain('**Bounded closer.**');
|
|
||||||
// Telemetry tail stripped: a hermetic child must not write to the
|
|
||||||
// operator's real ~/.gstack.
|
|
||||||
expect(skill).not.toContain('Eureka');
|
|
||||||
// Extraction proof: none of ship's workflow rode along.
|
|
||||||
expect(skill).not.toContain('## Preamble (run first)');
|
|
||||||
expect(skill).not.toContain('Review Readiness');
|
|
||||||
expect(skill.length).toBeLessThan(8192);
|
|
||||||
});
|
|
||||||
|
|
||||||
test('with-arm installs the skill + routing line; without-arm installs neither; both get git + bare origin', () => {
|
|
||||||
const withArm = setupArm(TASKS[0], 'with-skill');
|
|
||||||
const withoutArm = setupArm(TASKS[0], 'without-skill');
|
|
||||||
try {
|
|
||||||
const skillPath = path.join(withArm.dir, '.claude', 'skills', SKILL_NAME, 'SKILL.md');
|
|
||||||
expect(fs.existsSync(skillPath)).toBe(true);
|
|
||||||
expect(fs.readFileSync(path.join(withArm.dir, 'CLAUDE.md'), 'utf-8')).toContain('## Skill routing');
|
|
||||||
|
|
||||||
expect(fs.existsSync(path.join(withoutArm.dir, '.claude'))).toBe(false);
|
|
||||||
expect(fs.readFileSync(path.join(withoutArm.dir, 'CLAUDE.md'), 'utf-8')).not.toContain('Skill routing');
|
|
||||||
|
|
||||||
// Both arms: seeded commit + working bare origin (merge-base-style
|
|
||||||
// commands must work inside the arm).
|
|
||||||
for (const arm of [withArm, withoutArm]) {
|
|
||||||
expect(run('git', ['rev-parse', 'HEAD'], arm.dir).trim()).toMatch(/^[0-9a-f]{40}$/);
|
|
||||||
expect(run('git', ['remote', 'get-url', 'origin'], arm.dir).trim()).toBe(arm.originDir);
|
|
||||||
expect(run('git', ['merge-base', 'origin/main', 'HEAD'], arm.dir).trim()).toMatch(/^[0-9a-f]{40}$/);
|
|
||||||
}
|
|
||||||
} finally {
|
|
||||||
for (const arm of [withArm, withoutArm]) {
|
|
||||||
fs.rmSync(arm.dir, { recursive: true, force: true });
|
|
||||||
fs.rmSync(arm.originDir, { recursive: true, force: true });
|
|
||||||
}
|
|
||||||
}
|
|
||||||
});
|
|
||||||
|
|
||||||
test('diff capture: stat parsing + a real zero-diff and non-zero-diff round trip', () => {
|
|
||||||
expect(parseDiffStat(' 3 files changed, 120 insertions(+), 4 deletions(-)\n'))
|
|
||||||
.toEqual({ filesChanged: 3, insertions: 120, deletions: 4, net: 116 });
|
|
||||||
expect(parseDiffStat(' 1 file changed, 2 insertions(+)\n'))
|
|
||||||
.toEqual({ filesChanged: 1, insertions: 2, deletions: 0, net: 2 });
|
|
||||||
expect(parseDiffStat(''))
|
|
||||||
.toEqual({ filesChanged: 0, insertions: 0, deletions: 0, net: 0 });
|
|
||||||
|
|
||||||
const arm = setupArm(TASKS[2], 'without-skill');
|
|
||||||
try {
|
|
||||||
// Zero-diff arm: a VALID cell, zeros across the board.
|
|
||||||
const clean = captureStagedDiff(arm.dir, arm.seedSha);
|
|
||||||
expect(clean.filesChanged).toBe(0);
|
|
||||||
expect(clean.net).toBe(0);
|
|
||||||
expect(clean.patch.trim()).toBe('');
|
|
||||||
|
|
||||||
// Modify + add a file: counts appear, patch carries the change.
|
|
||||||
fs.appendFileSync(path.join(arm.dir, 'README.md'), 'appended line\n');
|
|
||||||
fs.writeFileSync(path.join(arm.dir, 'new-file.txt'), 'one\ntwo\n');
|
|
||||||
const dirty = captureStagedDiff(arm.dir, arm.seedSha);
|
|
||||||
expect(dirty.filesChanged).toBe(2);
|
|
||||||
expect(dirty.insertions).toBe(3);
|
|
||||||
expect(dirty.deletions).toBe(0);
|
|
||||||
expect(dirty.net).toBe(3);
|
|
||||||
expect(dirty.patch).toContain('appended line');
|
|
||||||
} finally {
|
|
||||||
fs.rmSync(arm.dir, { recursive: true, force: true });
|
|
||||||
fs.rmSync(arm.originDir, { recursive: true, force: true });
|
|
||||||
}
|
|
||||||
});
|
|
||||||
|
|
||||||
test('judge prompt construction embeds the rubric, the ticket, and the reference diffs', () => {
|
|
||||||
const goodDiff = fs.readFileSync(path.join(FIXTURES, 'reference', 'good-diff.patch'), 'utf-8');
|
|
||||||
const badDiff = fs.readFileSync(path.join(FIXTURES, 'reference', 'bad-diff.patch'), 'utf-8');
|
|
||||||
for (const diff of [goodDiff, badDiff]) {
|
|
||||||
const prompt = buildArmJudgePrompt(TASKS[0].ticket, diff, 'pinned0000');
|
|
||||||
expect(prompt).toContain('<<<UNTRUSTED_DIFF_pinned0000>>>');
|
|
||||||
expect(prompt).toContain('<<<END_UNTRUSTED_DIFF_pinned0000>>>');
|
|
||||||
expect(prompt).toContain(diff);
|
|
||||||
expect(prompt).toContain(TASKS[0].ticket);
|
|
||||||
expect(prompt).toContain('0-3 scale');
|
|
||||||
expect(prompt).toContain('Coverage is NOT over-engineering');
|
|
||||||
expect(prompt).toContain('MUST name the specific class, function, file, or pattern');
|
|
||||||
expect(prompt).toContain('construct MUST be exactly "none"');
|
|
||||||
}
|
|
||||||
// The reference diffs are what the rubric anchors describe: the bad diff
|
|
||||||
// carries a hand-rolled widget replacing a native element, the good one
|
|
||||||
// uses the platform.
|
|
||||||
expect(badDiff).toContain('class CalendarWidget');
|
|
||||||
expect(goodDiff).toContain('type="date"');
|
|
||||||
|
|
||||||
// Injection hardening: without an explicit sentinel, each call gets its
|
|
||||||
// own random block markers — an arm diff cannot pre-write a closing
|
|
||||||
// marker it has never seen.
|
|
||||||
const a = buildArmJudgePrompt(TASKS[0].ticket, goodDiff);
|
|
||||||
const b = buildArmJudgePrompt(TASKS[0].ticket, goodDiff);
|
|
||||||
const marker = (p: string) => /<<<UNTRUSTED_DIFF_([a-z0-9]+)>>>/.exec(p)?.[1];
|
|
||||||
expect(marker(a)).toBeTruthy();
|
|
||||||
expect(marker(b)).toBeTruthy();
|
|
||||||
expect(marker(a)).not.toBe(marker(b));
|
|
||||||
});
|
|
||||||
|
|
||||||
test('judge response parsing: reference-shaped verdicts accepted, malformed rejected', () => {
|
|
||||||
// Canned verdicts the judge should return for the reference diffs.
|
|
||||||
const goodVerdict = parseArmJudgeResponse({
|
|
||||||
over_engineering: 0,
|
|
||||||
construct: 'none',
|
|
||||||
reasoning: 'Native date input with a min attribute; nothing unrequested.',
|
|
||||||
});
|
|
||||||
expect(goodVerdict.over_engineering).toBe(0);
|
|
||||||
expect(goodVerdict.construct).toBe('none');
|
|
||||||
|
|
||||||
const badVerdict = parseArmJudgeResponse({
|
|
||||||
over_engineering: 3,
|
|
||||||
construct: 'hand-rolled CalendarWidget + DatePickerFactory in calendar.js',
|
|
||||||
reasoning: 'A custom calendar widget layer replaces <input type="date">.',
|
|
||||||
});
|
|
||||||
expect(badVerdict.over_engineering).toBe(3);
|
|
||||||
expect(badVerdict.construct).toContain('CalendarWidget');
|
|
||||||
|
|
||||||
// Malformed shapes throw — that throw is what the bounded retry catches.
|
|
||||||
expect(() => parseArmJudgeResponse({ over_engineering: 7, construct: 'x' })).toThrow(/integer 0-3/);
|
|
||||||
expect(() => parseArmJudgeResponse({ over_engineering: 1.5, construct: 'x' })).toThrow(/integer 0-3/);
|
|
||||||
expect(() => parseArmJudgeResponse({ over_engineering: 2 })).toThrow(/construct missing/);
|
|
||||||
expect(() => parseArmJudgeResponse({ over_engineering: 2, construct: 'none' })).toThrow(/must name the specific construct/);
|
|
||||||
expect(() => parseArmJudgeResponse({ over_engineering: 0, construct: 'a helper' })).toThrow(/construct "none"/);
|
|
||||||
expect(() => parseArmJudgeResponse(null)).toThrow();
|
|
||||||
});
|
|
||||||
|
|
||||||
test('armJudge: zero diff scores deterministically as none with no API call', async () => {
|
|
||||||
// No ANTHROPIC client is ever constructed on this path — safe keyless.
|
|
||||||
const score = await armJudge(TASKS[0].ticket, ' \n');
|
|
||||||
expect(score.over_engineering).toBe(0);
|
|
||||||
expect(score.construct).toBe('none');
|
|
||||||
});
|
|
||||||
|
|
||||||
test('armJudge: bounded retry-on-malformed — recovers once, then gives up', async () => {
|
|
||||||
// Malformed first, valid second: recovers within the 2-attempt bound.
|
|
||||||
let calls = 0;
|
|
||||||
const flaky = (async () => {
|
|
||||||
calls++;
|
|
||||||
return calls === 1
|
|
||||||
? { over_engineering: 9, construct: 'garbage' }
|
|
||||||
: { over_engineering: 2, construct: 'repository layer in app.js', reasoning: 'ok' };
|
|
||||||
}) as unknown as typeof callJudge;
|
|
||||||
const recovered = await armJudge('ticket', 'diff --git a/x b/x\n+1\n', { call: flaky });
|
|
||||||
expect(recovered.over_engineering).toBe(2);
|
|
||||||
expect(calls).toBe(ARM_JUDGE_ATTEMPTS);
|
|
||||||
|
|
||||||
// Always malformed: throws after exactly ARM_JUDGE_ATTEMPTS attempts.
|
|
||||||
let badCalls = 0;
|
|
||||||
const alwaysBad = (async () => {
|
|
||||||
badCalls++;
|
|
||||||
return { nonsense: true };
|
|
||||||
}) as unknown as typeof callJudge;
|
|
||||||
await expect(armJudge('ticket', 'diff --git a/x b/x\n+1\n', { call: alwaysBad }))
|
|
||||||
.rejects.toThrow(/no well-formed verdict after 2 attempts/);
|
|
||||||
expect(badCalls).toBe(ARM_JUDGE_ATTEMPTS);
|
|
||||||
});
|
|
||||||
});
|
|
||||||
|
|||||||
Reference in New Issue
Block a user