From c9e9a653cf6d4203d848b3e25289fc4bbd4ed7d4 Mon Sep 17 00:00:00 2001 From: Garry Tan Date: Sat, 29 Aug 2026 05:57:56 +0000 Subject: [PATCH] =?UTF-8?q?feat(evals):=20arm=20benchmark=20runs=20each=20?= =?UTF-8?q?fixture's=20functional=20oracle=20=E2=80=94=20correctness=20bef?= =?UTF-8?q?ore=20LOC?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The plan's metric order is diff-quality FIRST, but cells never ran the fixtures' own run-tests.js, so a refusal, a broken implementation, and working code were indistinguishable in aggregates (Codex adversarial catch). Tasks with an oracle declare checkCmd; every cell records checks=pass|fail|none in the report line and eval store. Selftest pins the oracle declarations and that the planted bug fails its own check pre-fix. Co-Authored-By: Claude Fable 5 --- test/arm-benchmark-selftest.test.ts | 29 ++++++++++++++++++++++++++- test/helpers/arm-benchmark-harness.ts | 16 +++++++++++++++ test/skill-e2e-arm-benchmark.test.ts | 14 ++++++++++--- 3 files changed, 55 insertions(+), 4 deletions(-) diff --git a/test/arm-benchmark-selftest.test.ts b/test/arm-benchmark-selftest.test.ts index 070b062cb..e622bf645 100644 --- a/test/arm-benchmark-selftest.test.ts +++ b/test/arm-benchmark-selftest.test.ts @@ -9,7 +9,7 @@ import { describe, test, expect } from 'bun:test'; import { TASKS, FIXTURES, SKILL_NAME, - buildBehavioralSkill, run, setupArm, parseDiffStat, captureStagedDiff, + buildBehavioralSkill, run, setupArm, parseDiffStat, captureStagedDiff, runChecks, } from './helpers/arm-benchmark-harness'; import { armJudge, buildArmJudgePrompt, parseArmJudgeResponse, @@ -206,3 +206,30 @@ describe('arm benchmark selftest (free, no API)', () => { expect(badCalls).toBe(ARM_JUDGE_ATTEMPTS); }); }); + +describe('functional checks (correctness before LOC)', () => { + test('every fixture with a run-tests.js oracle declares checkCmd; the trap fixtures behave as planted', () => { + for (const task of TASKS) { + const oracle = path.join(FIXTURES, task.fixture, 'run-tests.js'); + if (fs.existsSync(oracle)) { + expect(task.checkCmd, `${task.key} has run-tests.js but no checkCmd — its cells would report checks=none`).toEqual(['node', 'run-tests.js']); + } else { + expect(task.checkCmd).toBeUndefined(); + } + } + + // Pre-fix, the bugfix fixture MUST fail its own oracle (the planted bug), + // and a task with no oracle reports 'none' — never a throw. + const arm = setupArm(TASKS[2], 'without-skill'); + const noOracle = setupArm(TASKS[0], 'without-skill'); + try { + expect(runChecks(TASKS[2], arm.dir)).toBe('fail'); + expect(runChecks(TASKS[0], noOracle.dir)).toBe('none'); + } finally { + for (const a of [arm, noOracle]) { + fs.rmSync(a.dir, { recursive: true, force: true }); + fs.rmSync(a.originDir, { recursive: true, force: true }); + } + } + }); +}); diff --git a/test/helpers/arm-benchmark-harness.ts b/test/helpers/arm-benchmark-harness.ts index d122614ea..33fcfcef4 100644 --- a/test/helpers/arm-benchmark-harness.ts +++ b/test/helpers/arm-benchmark-harness.ts @@ -38,6 +38,10 @@ export interface ArmTask { /** E2E_TOUCHFILES / E2E_TIERS key. Quoted literals below keep the parent * shard mapper (test-paid-shards) able to attribute this file. */ key: string; + /** Deterministic functional check run against the arm's working tree after + * harvest (the plan's metric order is diff-quality FIRST: correctness + * before LOC). Absent for fixtures with no runnable oracle. */ + checkCmd?: string[]; fixture: string; ticket: string; } @@ -58,6 +62,7 @@ Leave your changes uncommitted in the working tree.`, { key: 'arm-benchmark-crud-endpoint', fixture: 'crud-endpoint', + checkCmd: ['node', 'run-tests.js'], ticket: `Ticket: users need to delete notes. You are in a small git repo containing an in-memory notes API (app.js, wired to HTTP in server.js). @@ -68,6 +73,7 @@ Leave your changes uncommitted in the working tree.`, { key: 'arm-benchmark-bugfix-decoys', fixture: 'bugfix-decoys', + checkCmd: ['node', 'run-tests.js'], ticket: `Bug report: receipts print $10.5 for a $10.05 item. You are in a small git repo. \`node run-tests.js\` currently fails on formatPrice(1005). @@ -217,3 +223,13 @@ export function captureStagedDiff(dir: string, seedSha: string): DiffHarvest { return { ...parseDiffStat(stat), stat: stat.trim(), patch }; } + +/** Run the task's functional check in the arm dir. 'none' when the task has + * no oracle; never throws — a crashing check is a 'fail', not a dead cell. */ +export function runChecks(task: ArmTask, dir: string): 'pass' | 'fail' | 'none' { + if (!task.checkCmd || task.checkCmd.length === 0) return 'none'; + const r = spawnSync(task.checkCmd[0], task.checkCmd.slice(1), { + cwd: dir, stdio: 'pipe', encoding: 'utf-8', timeout: 60_000, + }); + return r.status === 0 ? 'pass' : 'fail'; +} diff --git a/test/skill-e2e-arm-benchmark.test.ts b/test/skill-e2e-arm-benchmark.test.ts index d74f781b7..01e3d7ee9 100644 --- a/test/skill-e2e-arm-benchmark.test.ts +++ b/test/skill-e2e-arm-benchmark.test.ts @@ -45,7 +45,7 @@ import { armJudge, type ArmJudgeScore } from './helpers/llm-judge'; import { ARM_MAX_TURNS, ARM_TIMEOUT_MS, ARM_JUDGE_DIFF_CAP, ARM_ALLOWED_TOOLS, TASK_TEST_TIMEOUT_MS, SKILL_NAME, TASKS, - setupArm, captureStagedDiff, + setupArm, captureStagedDiff, runChecks, type Arm, type ArmTask, type DiffHarvest, } from './helpers/arm-benchmark-harness'; import * as fs from 'fs'; @@ -61,6 +61,10 @@ interface CellResult { harvestError: string | null; judge: ArmJudgeScore | null; judgeError: string | null; + /** Deterministic functional-check outcome ('none' = task has no oracle). + * Correctness comes before LOC in the metric order — a refusal, a broken + * implementation, and working code must be distinguishable in the cells. */ + checks: 'pass' | 'fail' | 'none'; consulted: boolean; costUsd: number; tokens: number; @@ -102,6 +106,7 @@ async function runArmCell(task: ArmTask, arm: Arm): Promise { } catch (err) { harvestError = err instanceof Error ? err.message : String(err); } + const checks = runChecks(task, dirs.dir); // Judge taxonomy: still malformed after armJudge's bounded retries -> // judge_error cell (excluded from aggregates, surfaced in the report). @@ -131,7 +136,9 @@ async function runArmCell(task: ArmTask, arm: Arm): Promise { net: harvest.net, } : null, - judge_scores: judge ? { over_engineering: judge.over_engineering } : undefined, + judge_scores: judge + ? { over_engineering: judge.over_engineering, ...(checks !== 'none' ? { checks_pass: checks === 'pass' ? 1 : 0 } : {}) } + : undefined, judge_reasoning: judge ? `construct: ${judge.construct} | ${judge.reasoning}${judgeDiffTruncated ? ` | diff truncated to ${ARM_JUDGE_DIFF_CAP}B` : ''}` : judgeError ? `judge_error: ${judgeError}` : undefined, @@ -146,6 +153,7 @@ async function runArmCell(task: ArmTask, arm: Arm): Promise { harvestError, judge, judgeError, + checks, consulted, costUsd: result.costEstimate.estimatedCost, tokens: result.costEstimate.estimatedTokens, @@ -166,7 +174,7 @@ function cellLine(c: CellResult): string { const loc = c.harvest ? `+${c.harvest.insertions}/-${c.harvest.deletions} net ${c.harvest.net} in ${c.harvest.filesChanged} file(s)` : `harvest FAILED: ${c.harvestError}`; - return ` ${c.arm.padEnd(14)} score=${score} loc=${loc} turns=${c.turns} ` + return ` ${c.arm.padEnd(14)} score=${score} checks=${c.checks} loc=${loc} turns=${c.turns} ` + `tokens=${(c.tokens / 1000).toFixed(1)}k cost=$${c.costUsd.toFixed(2)} consulted=${c.consulted}`; }