v1.90.0.0 feat: make browser cookie imports explicit and safe (#2964)

* fix(browse): prepare reliable cookie import wave for validation

* ci: sequence quality and behavior for validation branch

* fix(browse): isolate Windows qualification and preserve native diagnostics

* test(browse): cover cookie workflow quality and isolate Windows user paths

* test(browse): trace native member startup and initialize fresh folders

* fix(browse): keep Windows member stdin alive through EOF

* fix(browse): latch native timeouts and compare contained Edge startup

* test(browse): verify native version metadata and actual Windows argv

* test(browse): qualify Dia import on isolated macOS CI

* fix(browse): require picker origin for session mutations

* fix(browse): bound credential reads through stream completion

* test(browse): inspect owned Windows process arguments natively

* test(evals): preserve passing coverage during cookie repair reruns

* test(browse): isolate Dia qualification in a fresh macOS account

* test(browse): pass bounded integer timeouts to native Mac probes

* test(browse): distinguish Windows profile initialization from containment

* test(browse): await descendant pipe readiness before parent exit

* test(browse): initialize and restore isolated macOS Keychain state

* test(browse): initialize Windows fixture folders before qualification

* test(ci): pin the same Node runtime across Windows checks

* test(browse): distinguish native macOS browser preflight stages

* test(browse): isolate Windows descendant console lifetime

* test(browse): preserve native receipts and identify fixture lock holders

* test(browse): prepare dependency resolution before native Mac worker startup

* test(ci): include lock and close checks in native diagnostics

* test(browse): preserve native owner probe stages and subprocess deadlines

* fix(browse): classify Chromium profile-in-use exit precisely

* test(browse): retain Mac qualification evidence through cleanup failures

* test(browse): bound Mac fixture paths and retire its owned user domain

* test(browse): accept vanished fixture entries without weakening cleanup

* test(browse): identify probe-created macOS user domains safely

* test(browse): observe Mac user domains without targeting them first

* test(browse): use passive fresh-user ownership throughout Mac qualification

* test(browse): distinguish profile and registered-home Keychain lookups

* test(browse): qualify Dia under one registered account home

* test(browse): identify Dia startup and owned process-group failures

* test(browse): classify bounded Dia startup diagnostics without leaking output

* fix(test): preserve native Mac sandboxing and reap owned browser children

* fix(browse): preserve Chromium sandboxing for native profile imports

* test(browse): inspect signed Mach-O architecture without launching Xcode tools

* test(browse): sample pending Dia startup and reap on all cleanup paths

* test(browse): compare protected Dia launches in fresh Bun and Node accounts

* test(browse): inspect isolated Mac GUI readiness without browser access

* v1.90.0.0 fix: bind cookie picker actions to their document

* test: validate cookie guards and fit nested launch fixtures

* ci: configure the bundled Chromium sandbox helper

* fix(browse): classify Playwright authentication timeouts

* test: retain bounded Windows lifecycle diagnostics

* test(cso): reuse bounded NTFS precision candidates

* test(review): handle explicit preservation choices safely

* test(browse): remove owned fixture directories with explicit primitives

* test(review): distinguish descriptive reuse from edit commitments

* test: admit only the approved unscored cookie workflow refusal

* test: keep the Office Hours judge mock export-complete

* fix: keep dependency-free CI planners independent of the model SDK

* test: observe the exact holder after a native fixture unlink failure

* fix: start seeded PTY observations at owned readiness

* test: acquire identity-bound Windows deletion admission before profile resets

* test: preserve qualified Git index bits without authorizing mutations
This commit is contained in:
Garry Tan
2026-09-25 12:06:45 -04:00
committed by GitHub
parent 730a1017d1
commit a84b0b5b6d
111 changed files with 14996 additions and 1057 deletions
+1 -1
View File
@@ -4605,7 +4605,7 @@ export async function runPlanSkillObservation(opts: {
};
// Entry deadline → boot → owned paste/receipt/ack → slash → observation.
// Setup consumes the existing case budget; cleanup has its separate grace.
await Bun.sleep(Math.min(8000, Math.max(0, deadlineAt - Date.now())));
if (!opts.initialPlanContent) await Bun.sleep(Math.min(8000, Math.max(0, deadlineAt - Date.now())));
if (opts.initialPlanContent) {
const seed = `Keep this draft plan as context. Briefly acknowledge receipt, then wait for my next message containing a slash command. Do not start the review or call tools yet.\n\n${opts.initialPlanContent}`;
try {
@@ -0,0 +1,50 @@
import { createHash } from 'node:crypto';
import { readFileSync } from 'node:fs';
import { join } from 'node:path';
import { buildWorkflowJudgePrompt, type WorkflowJudgeFile, type WorkflowJudgeInput } from './workflow-judge-input';
export const COOKIE_WORKFLOW_JUDGE = {
judgeContext: 'a fallback-browser cookie import workflow',
judgeGoal: 'how to select an authorized source browser, profile, and domain without guessing an account; configure optional authentication verification before mutation; obtain explicit consent for precisely scoped storage reset; distinguish copied cookies from positive sign-in evidence; and recover within the documented platform and privacy boundaries',
thresholds: { clarity: 4, completeness: 3, actionability: 4 },
} as const;
export function buildCookieWorkflowJudgeInput(root: string): WorkflowJudgeInput & { prompt: string; sha256: string } {
const files: WorkflowJudgeFile[] = [
{ path: 'setup-browser-cookies/SKILL.md', kind: 'entrypoint', start: '# Setup Browser Cookies', end: null },
{ path: 'BROWSER.md', kind: 'section', start: '#### Choosing a source and checking sign-in', end: '### Tabs + frames' },
].map(spec => {
const source = readFileSync(join(root, spec.path), 'utf8');
const locate = (marker: string): number => {
const matches = [...source.matchAll(new RegExp(`^${marker.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}\\r?$`, 'gm'))];
if (matches.length !== 1) throw new Error(`${spec.path}: expected exactly one marker ${JSON.stringify(marker)}, found ${matches.length}`);
return matches[0].index!;
};
const start = locate(spec.start);
const end = spec.end === null ? source.length : locate(spec.end);
if (end <= start || !source.slice(start + spec.start.length, end).trim()) {
throw new Error(`${spec.path}: empty or reversed cookie workflow excerpt`);
}
return {
path: spec.path,
kind: spec.kind as WorkflowJudgeFile['kind'],
content: source.slice(start, end),
startLine: source.slice(0, start).split('\n').length,
endLine: source.slice(0, end - 1).split('\n').length,
};
});
if (!files[0].content.includes('`BROWSER.md`') || !files[0].content.includes('**Choosing a source and checking sign-in**')) {
throw new Error('setup-browser-cookies/SKILL.md: missing cookie reference link');
}
const text = [
'SKILL.md is the entry point. BROWSER.md supplies the exact referenced cookie section, not an additional execution step.',
...files.map(file => [
`--- BEGIN FILE ${JSON.stringify(file.path)} (lines ${file.startLine}-${file.endLine}; ${file.kind}) ---`,
file.content,
`--- END FILE ${JSON.stringify(file.path)} ---`,
].join('\n')),
].join('\n\n');
const input = { files, text };
const prompt = buildWorkflowJudgePrompt(COOKIE_WORKFLOW_JUDGE, input);
return { ...input, prompt, sha256: createHash('sha256').update(prompt).digest('hex') };
}
@@ -0,0 +1,101 @@
import { createHash } from 'node:crypto';
import { readFileSync } from 'node:fs';
import { join } from 'node:path';
import { buildCookieWorkflowJudgeInput, COOKIE_WORKFLOW_JUDGE } from './cookie-workflow-judge-input';
import type { JudgeRefusalEvidence } from './llm-judge';
import { DEFAULT_JUDGE_MAX_TOKENS, resolveEvalModel } from '../../lib/eval-model';
export const COOKIE_MANUAL_REVIEW_FILE = '.github/cookie-workflow-manual-review.json';
const CASE = 'setup-browser-cookies/SKILL.md workflow';
type Thresholds = { clarity: number; completeness: number; actionability: number };
export interface CookieManualApproval {
schema_version: 1;
test_name: typeof CASE;
prompt_sha256: string;
prompt_bytes: number;
model: string;
max_tokens: number;
thresholds: Thresholds;
approved_by: string;
approved_at: string;
approval_url: string;
reason: string;
}
export interface ManualJudgeReview {
approval: CookieManualApproval;
refusal: JudgeRefusalEvidence;
}
const object = (value: unknown): value is Record<string, any> => value !== null && typeof value === 'object' && !Array.isArray(value);
const nonempty = (value: unknown): value is string => typeof value === 'string' && value.trim().length > 0;
const dimensions = ['clarity', 'completeness', 'actionability'] as const;
const sameThresholds = (a: Thresholds, b: Thresholds) => dimensions.every(key => a[key] === b[key]);
function validApproval(value: unknown): value is CookieManualApproval {
return object(value) && value.schema_version === 1 && value.test_name === CASE
&& typeof value.prompt_sha256 === 'string' && /^[a-f0-9]{64}$/.test(value.prompt_sha256)
&& Number.isSafeInteger(value.prompt_bytes) && value.prompt_bytes > 0
&& nonempty(value.model) && Number.isSafeInteger(value.max_tokens) && value.max_tokens > 0
&& object(value.thresholds) && dimensions.every(key => Number.isInteger(value.thresholds[key]) && value.thresholds[key] >= 1 && value.thresholds[key] <= 5)
&& nonempty(value.approved_by) && typeof value.approved_at === 'string' && /^\d{4}-\d{2}-\d{2}$/.test(value.approved_at)
&& typeof value.approval_url === 'string' && /^https:\/\/github\.com\/garrytan\/gstack\/pull\/2964#issuecomment-\d+$/.test(value.approval_url)
&& nonempty(value.reason);
}
function validRefusal(value: unknown, model: string): value is JudgeRefusalEvidence {
return object(value) && value.stop_reason === 'refusal' && value.model === model
&& nonempty(value.response_id) && nonempty(value.request_id)
&& Number.isSafeInteger(value.input_tokens) && value.input_tokens >= 0
&& value.output_tokens === 0 && value.text_blocks === 0;
}
export function isManualReviewEntry(entry: unknown): boolean {
if (!object(entry) || entry.name !== CASE || entry.suite !== 'Cookie setup workflow quality'
|| entry.tier !== 'llm-judge' || entry.passed !== false
|| entry.attempt !== 1
|| entry.execution !== 'executed' || entry.exit_reason !== 'provider_refusal'
|| entry.judge_scores !== undefined || entry.judge_reasoning !== undefined || entry.reused_from !== undefined
|| typeof entry.prompt !== 'string' || !object(entry.manual_review)) return false;
const { approval, refusal } = entry.manual_review;
return validApproval(approval) && entry.model === approval.model && validRefusal(refusal, approval.model)
&& Buffer.byteLength(entry.prompt) === approval.prompt_bytes
&& createHash('sha256').update(entry.prompt).digest('hex') === approval.prompt_sha256;
}
export function getCookieWorkflowManualReview(root: string, request: {
testName: string; prompt: string; model: string; maxTokens: number; thresholds: Thresholds; attempt: number;
}, refusal: JudgeRefusalEvidence): ManualJudgeReview | null {
if (request.testName !== CASE || request.attempt !== 1 || !validRefusal(refusal, request.model)) return null;
let approval: unknown;
try { approval = JSON.parse(readFileSync(join(root, COOKIE_MANUAL_REVIEW_FILE), 'utf8')); }
catch (error) {
if ((error as NodeJS.ErrnoException).code === 'ENOENT') return null;
throw error;
}
if (!validApproval(approval)) throw new Error('Invalid cookie workflow manual-review approval');
const current = buildCookieWorkflowJudgeInput(root);
if (request.prompt !== current.prompt || current.sha256 !== approval.prompt_sha256
|| Buffer.byteLength(request.prompt) !== approval.prompt_bytes || request.model !== approval.model
|| request.maxTokens !== approval.max_tokens || request.maxTokens !== DEFAULT_JUDGE_MAX_TOKENS
|| !sameThresholds(request.thresholds, approval.thresholds)
|| !sameThresholds(request.thresholds, COOKIE_WORKFLOW_JUDGE.thresholds)) return null;
return { approval, refusal };
}
export function manualReviewProblem(entry: unknown, root: string): string | null {
if (!object(entry) || !Object.hasOwn(entry, 'manual_review')) return null;
if (!isManualReviewEntry(entry)) return 'Malformed manual-review claim';
if (entry.model !== resolveEvalModel('judge')) return 'Manual review model is not the current judge model';
try {
const claimed = entry.manual_review as ManualJudgeReview;
const verified = getCookieWorkflowManualReview(root, { testName: entry.name, prompt: entry.prompt,
model: entry.model, maxTokens: claimed.approval.max_tokens, thresholds: claimed.approval.thresholds,
attempt: entry.attempt }, claimed.refusal);
if (!verified || Object.entries(verified.approval).some(([key, value]) => key === 'thresholds'
? !sameThresholds(value as Thresholds, claimed.approval.thresholds)
: value !== claimed.approval[key as keyof CookieManualApproval])) return 'Manual review does not match current source and approval';
return null;
} catch { return 'Manual review approval or current input is unavailable or invalid'; }
}
+34
View File
@@ -0,0 +1,34 @@
import * as fs from 'node:fs';
import * as path from 'node:path';
export function createPrecisionLossCandidate(
target: string,
contents: string,
readFileId: (file: string) => bigint = file => fs.lstatSync(file, { bigint: true }).ino,
): bigint {
if (fs.existsSync(target)) throw new Error('Precision fixture target already exists');
const directory = path.dirname(target);
if (fs.lstatSync(directory).isSymbolicLink()) throw new Error('Precision fixture directory must not be linked');
const pool = fs.mkdtempSync(path.join(directory, 'ntfs-id-'));
let aboveSafe = 0;
try {
for (let batch = 0; batch < 512; batch++) {
const files: string[] = [];
for (let slot = 0; slot < 2; slot++) {
const file = path.join(pool, `candidate-${slot}`);
fs.writeFileSync(file, contents, { flag: 'wx' });
files.push(file);
const inode = readFileId(file);
if (inode > BigInt(Number.MAX_SAFE_INTEGER)) aboveSafe++;
if (inode > BigInt(Number.MAX_SAFE_INTEGER) && String(Number(inode)) !== String(inode)) {
fs.linkSync(file, target);
return inode;
}
}
for (const file of files) fs.unlinkSync(file);
}
throw new Error(`No precision-losing NTFS file ID within 1024 file creations (${aboveSafe} above-safe observations)`);
} finally {
fs.rmSync(pool, { recursive: true, force: true });
}
}
+1 -1
View File
@@ -107,7 +107,7 @@ export const FILE_RETRY_BUDGETS = [
...[
// Fourteen workflow judges include their 10s recording grace; the other
// eleven judges retain 120s. Supervise all 25 and the existing one retry.
{ file: 'test/skill-llm-eval.test.ts', attemptMs: 14 * (JUDGE_MS + 10_000) + 11 * JUDGE_MS, retries: 1 },
{ file: 'test/skill-llm-eval.test.ts', attemptMs: 15 * (JUDGE_MS + 10_000) + 11 * JUDGE_MS, retries: 1 },
{ file: 'test/codex-e2e-plan-format.test.ts', attemptMs: 4 * (CAPTURE_LONG_MS + 10_000), retries: 1 },
{ file: 'test/skill-e2e-auq-matrix.test.ts', attemptMs: 6 * CAPTURE_MS, retries: 1 },
{ file: 'test/skill-e2e-plan-format.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), retries: 1 },
+71
View File
@@ -10,11 +10,13 @@ import {
isPartialEval,
listEvalJsonFiles,
compareEvalResults,
evalEntryOutcome,
formatComparison,
generateCommentary,
judgePassed,
} from './eval-store';
import type { EvalResult, EvalTestEntry, ComparisonResult } from './eval-store';
import { manualReviewFixture } from './manual-judge-review-fixture';
let tmpDir: string;
@@ -77,6 +79,36 @@ async function captureStderr(fn: () => Promise<void>): Promise<string> {
// --- EvalCollector tests ---
describe('EvalCollector', () => {
test('manual provider refusal stays unscored and executed; malformed claims stay failed', async () => {
const manual = manualReviewFixture();
const invalidPass = { ...manual, passed: true };
const invalidScore = { ...manual, judge_scores: { clarity: 5 } };
expect(evalEntryOutcome(manual)).toBe('manual-review');
expect(evalEntryOutcome(invalidPass)).toBe('failed');
expect(evalEntryOutcome(invalidScore)).toBe('failed');
expect(evalEntryOutcome({ ...manual, manual_review: { ...manual.manual_review, approval: { ...manual.manual_review!.approval,
prompt_sha256: '0'.repeat(64) } } })).toBe('failed');
const collector = new EvalCollector('llm-judge', tmpDir);
collector.addTest(makeEntry({ name: 'ordinary', tier: 'llm-judge' }));
collector.addTest(manual);
collector.addTest(invalidPass);
collector.addTest(invalidScore);
const partial: EvalResult = JSON.parse(fs.readFileSync(path.join(tmpDir, '_partial-e2e.json'), 'utf8'));
expect(partial).toMatchObject({ passed: 1, failed: 2, manual_accepted_tests: 1,
executed_tests: 4, reused_tests: 0 });
const output = await captureStderr(async () => { await collector.finalize(); });
const result: EvalResult = JSON.parse(fs.readFileSync(fs.readdirSync(tmpDir).map(name => path.join(tmpDir, name))
.find(name => name.endsWith('.json') && !path.basename(name).startsWith('_'))!, 'utf8'));
expect(result).toMatchObject({ passed: 1, failed: 2, manual_accepted_tests: 1, executed_tests: 4 });
expect(result.tests[1]).toMatchObject({ passed: false, execution: 'executed', exit_reason: 'provider_refusal' });
expect(result.tests[1].judge_scores).toBeUndefined();
expect(output).toContain('MANUAL');
expect(output).toContain('1 unscored provider refusal');
expect(output).toContain(manual.manual_review!.approval.approval_url);
expect(result.tests[2].passed).toBe(true);
expect(output).toContain(' FAIL ');
});
test('reused passing evidence preserves origin and remains separate from newly executed attempts', async () => {
const collector = new EvalCollector('llm-judge', tmpDir);
const reused_from = { input_key: 'a'.repeat(64), run_id: '1234/1', revision: 'b'.repeat(40),
@@ -590,6 +622,45 @@ describe('findLatestFinalizedRun', () => {
// --- compareEvalResults tests ---
describe('compareEvalResults', () => {
test('manual acceptance is neither a score regression nor a recovery', () => {
const manual = manualReviewFixture();
const prior = makeResult({ tests: [makeEntry({ name: manual.name, passed: true })] });
const accepted = makeResult({ tests: [manual], passed: 0, failed: 0, manual_accepted_tests: 1 });
const toManual = compareEvalResults(prior, accepted, 'prior.json', 'accepted.json');
expect(toManual).toMatchObject({ improved: 0, regressed: 0, manual_reviewed: 1 });
expect(toManual.deltas[0]).toMatchObject({ status_change: 'manual-review', before: { passed: true },
after: { passed: false, manual_review: true } });
expect(formatComparison(toManual)).toContain('PASS → MANUAL');
expect(formatComparison(toManual)).not.toContain('REGRESSION:');
const fromManual = compareEvalResults(accepted, prior, 'accepted.json', 'prior.json');
expect(fromManual).toMatchObject({ improved: 0, regressed: 0, manual_reviewed: 1 });
expect(formatComparison(fromManual)).toContain('MANUAL → PASS');
const invalid = makeResult({ tests: [{ ...manual, passed: true }] });
expect(compareEvalResults(prior, invalid, 'prior.json', 'invalid.json').regressed).toBe(1);
});
test('manual acceptance followed by a real or malformed failure is a blocking regression', () => {
const manual = manualReviewFixture();
const accepted = makeResult({ tests: [manual], passed: 0, failed: 0, manual_accepted_tests: 1 });
const { manual_review: _receipt, ...ordinary } = manual;
for (const after of [
{ ...ordinary, exit_reason: 'timeout' },
{ ...manual, passed: true },
{ ...manual, judge_scores: { clarity: 5 } },
]) {
const failed = makeResult({ tests: [after] });
const comparison = compareEvalResults(accepted, failed, 'accepted.json', 'failed.json');
expect(comparison).toMatchObject({ improved: 0, regressed: 1 });
expect(comparison.manual_reviewed).toBeUndefined();
expect(comparison.deltas[0]).toMatchObject({ status_change: 'regressed', before: { manual_review: true },
after: { passed: false } });
const output = formatComparison(comparison);
expect(output).toContain('MANUAL → FAIL');
expect(output).toContain('REGRESSION:');
expect(output).toContain('blocking failure');
}
});
test('detects improved/regressed/unchanged per test', () => {
const before = makeResult({
tests: [
+57 -18
View File
@@ -12,6 +12,8 @@ import * as fs from 'fs';
import * as path from 'path';
import * as os from 'os';
import { spawnSync } from 'child_process';
import { isManualReviewEntry } from './cookie-workflow-manual-review';
import type { ManualJudgeReview } from './cookie-workflow-manual-review';
// v2: EvalTestEntry.harvest gains optional {insertions, deletions, net} and
// may be explicitly null (arm-benchmark harvest-failure taxonomy). Readers
@@ -66,6 +68,7 @@ export interface EvalTestEntry {
/** Absent in older records means executed; reuse is never a new model run. */
execution?: 'executed' | 'reused';
reused_from?: { input_key: string; run_id: string; revision: string; completed_at: string };
manual_review?: ManualJudgeReview;
/** 1-based record attempt for this name in this run. bun's --retry leaves
* retried passes INVISIBLE in its text output (a fail→pass prints no
* (fail) line and recaps as a clean pass — probed on 1.3.10), so the ONLY
@@ -120,6 +123,14 @@ export interface EvalTestEntry {
} | null;
}
export function evalEntryOutcome(entry: unknown): 'passed' | 'failed' | 'manual-review' {
if (!entry || typeof entry !== 'object') return 'failed';
if ('manual_review' in entry) return Object.hasOwn(entry, 'manual_review') && isManualReviewEntry(entry) ? 'manual-review' : 'failed';
const result = entry as EvalTestEntry;
if (result.execution !== undefined && result.execution !== 'executed' && result.execution !== 'reused') return 'failed';
return result.passed === true ? 'passed' : 'failed';
}
export interface EvalResult {
schema_version: number;
version: string;
@@ -135,6 +146,7 @@ export interface EvalResult {
total_tests: number;
executed_tests?: number;
reused_tests?: number;
manual_accepted_tests?: number;
passed: number;
failed: number;
total_cost_usd: number;
@@ -154,10 +166,10 @@ export interface EvalResult {
export interface TestDelta {
name: string;
before: { passed: boolean; cost_usd: number; turns_used?: number; duration_ms?: number;
detection_rate?: number; tool_summary?: Record<string, number> };
detection_rate?: number; tool_summary?: Record<string, number>; manual_review?: boolean };
after: { passed: boolean; cost_usd: number; turns_used?: number; duration_ms?: number;
detection_rate?: number; tool_summary?: Record<string, number> };
status_change: 'improved' | 'regressed' | 'unchanged';
detection_rate?: number; tool_summary?: Record<string, number>; manual_review?: boolean };
status_change: 'improved' | 'regressed' | 'unchanged' | 'manual-review';
}
export interface ComparisonResult {
@@ -173,6 +185,7 @@ export interface ComparisonResult {
improved: number;
regressed: number;
unchanged: number;
manual_reviewed?: number;
tool_count_before: number;
tool_count_after: number;
/** After-tests that had a same-named entry in the before run. 0 = nothing was
@@ -388,6 +401,7 @@ export function compareEvalResults(
): ComparisonResult {
const deltas: TestDelta[] = [];
let improved = 0, regressed = 0, unchanged = 0;
let manualReviewed = 0;
let toolCountBefore = 0, toolCountAfter = 0;
let matched = 0;
@@ -409,33 +423,40 @@ export function compareEvalResults(
toolCountAfter += afterToolCount;
let statusChange: TestDelta['status_change'] = 'unchanged';
const beforeManual = beforeTest !== undefined && evalEntryOutcome(beforeTest) === 'manual-review';
const afterOutcome = evalEntryOutcome(afterTest);
const afterManual = afterOutcome === 'manual-review';
if (beforeTest) {
matched++;
if (!beforeTest.passed && afterTest.passed) { statusChange = 'improved'; improved++; }
else if (beforeTest.passed && !afterTest.passed) { statusChange = 'regressed'; regressed++; }
if (beforeManual && afterOutcome === 'failed') { statusChange = 'regressed'; regressed++; }
else if (beforeManual || afterManual) { statusChange = 'manual-review'; manualReviewed++; }
else if (evalEntryOutcome(beforeTest) === 'failed' && evalEntryOutcome(afterTest) === 'passed') { statusChange = 'improved'; improved++; }
else if (evalEntryOutcome(beforeTest) === 'passed' && evalEntryOutcome(afterTest) === 'failed') { statusChange = 'regressed'; regressed++; }
else { unchanged++; }
} else {
// New test — treat as unchanged (no prior data)
unchanged++;
if (afterManual) { statusChange = 'manual-review'; manualReviewed++; }
else unchanged++;
}
deltas.push({
name: afterTest.name,
before: {
passed: beforeTest?.passed ?? false,
passed: beforeTest !== undefined && evalEntryOutcome(beforeTest) === 'passed',
cost_usd: beforeTest?.cost_usd ?? 0,
turns_used: beforeTest?.turns_used,
duration_ms: beforeTest?.duration_ms,
detection_rate: beforeTest?.detection_rate,
tool_summary: beforeToolSummary,
...(beforeManual ? { manual_review: true } : {}),
},
after: {
passed: afterTest.passed,
passed: afterOutcome === 'passed',
cost_usd: afterTest.cost_usd,
turns_used: afterTest.turns_used,
duration_ms: afterTest.duration_ms,
detection_rate: afterTest.detection_rate,
tool_summary: afterToolSummary,
...(afterManual ? { manual_review: true } : {}),
},
status_change: statusChange,
});
@@ -452,12 +473,13 @@ export function compareEvalResults(
deltas.push({
name: `${name} (removed)`,
before: {
passed: beforeTest.passed,
passed: evalEntryOutcome(beforeTest) === 'passed',
cost_usd: beforeTest.cost_usd,
turns_used: beforeTest.turns_used,
duration_ms: beforeTest.duration_ms,
detection_rate: beforeTest.detection_rate,
tool_summary: beforeToolSummary,
...(evalEntryOutcome(beforeTest) === 'manual-review' ? { manual_review: true } : {}),
},
after: { passed: false, cost_usd: 0, tool_summary: {} },
status_change: 'unchanged',
@@ -477,6 +499,7 @@ export function compareEvalResults(
improved,
regressed,
unchanged,
...(manualReviewed ? { manual_reviewed: manualReviewed } : {}),
tool_count_before: toolCountBefore,
tool_count_after: toolCountAfter,
matched,
@@ -495,8 +518,8 @@ export function formatComparison(c: ComparisonResult): string {
// Per-test deltas
for (const d of c.deltas) {
const arrow = d.status_change === 'improved' ? '↑' : d.status_change === 'regressed' ? '↓' : '=';
const beforeStatus = d.before.passed ? 'PASS' : 'FAIL';
const afterStatus = d.after.passed ? 'PASS' : 'FAIL';
const beforeStatus = d.before.manual_review ? 'MANUAL' : d.before.passed ? 'PASS' : 'FAIL';
const afterStatus = d.after.manual_review ? 'MANUAL' : d.after.passed ? 'PASS' : 'FAIL';
// Turns delta
let turnsDelta = '';
@@ -540,6 +563,7 @@ export function formatComparison(c: ComparisonResult): string {
if (c.improved > 0) parts.push(`${c.improved} improved`);
if (c.regressed > 0) parts.push(`${c.regressed} regressed`);
if (c.unchanged > 0) parts.push(`${c.unchanged} unchanged`);
if (c.manual_reviewed) parts.push(`${c.manual_reviewed} unscored manual review`);
lines.push(` Status: ${parts.join(', ')}`);
const costSign = c.total_cost_delta >= 0 ? '+' : '';
@@ -607,7 +631,9 @@ export function generateCommentary(c: ComparisonResult): string[] {
const regressions = c.deltas.filter(d => d.status_change === 'regressed');
if (regressions.length > 0) {
for (const d of regressions) {
notes.push(`REGRESSION: "${d.name}" was passing, now fails. Investigate immediately.`);
notes.push(d.before.manual_review
? `REGRESSION: "${d.name}" lost its unscored manual acceptance and now has a blocking failure. Investigate immediately.`
: `REGRESSION: "${d.name}" was passing, now fails. Investigate immediately.`);
}
}
@@ -617,6 +643,10 @@ export function generateCommentary(c: ComparisonResult): string[] {
notes.push(`Fixed: "${d.name}" now passes.`);
}
for (const d of c.deltas.filter(delta => delta.status_change === 'manual-review')) {
notes.push(`"${d.name}" includes an unscored manual acceptance; no model-score improvement or regression is inferred.`);
}
// 3. Per-test efficiency changes (only for unchanged-status tests — regressions/improvements are already noted)
const stable = c.deltas.filter(d => d.status_change === 'unchanged' && d.after.passed);
for (const d of stable) {
@@ -893,7 +923,8 @@ export class EvalCollector {
const version = getVersion();
const totalCost = this.tests.reduce((s, t) => s + t.cost_usd, 0);
const totalDuration = this.tests.reduce((s, t) => s + t.duration_ms, 0);
const passed = this.tests.filter(t => t.passed).length;
const passed = this.tests.filter(t => evalEntryOutcome(t) === 'passed').length;
const manual = this.tests.filter(t => evalEntryOutcome(t) === 'manual-review').length;
const partial: EvalResult = {
schema_version: SCHEMA_VERSION,
@@ -907,8 +938,9 @@ export class EvalCollector {
total_tests: this.tests.length,
executed_tests: this.tests.filter(t => t.execution !== 'reused').length,
reused_tests: this.tests.filter(t => t.execution === 'reused').length,
...(manual ? { manual_accepted_tests: manual } : {}),
passed,
failed: this.tests.length - passed,
failed: this.tests.length - passed - manual,
total_cost_usd: Math.round(totalCost * 100) / 100,
total_duration_ms: totalDuration,
tests: this.tests,
@@ -933,7 +965,8 @@ export class EvalCollector {
const timestamp = new Date().toISOString();
const totalCost = this.tests.reduce((s, t) => s + t.cost_usd, 0);
const totalDuration = this.tests.reduce((s, t) => s + t.duration_ms, 0);
const passed = this.tests.filter(t => t.passed).length;
const passed = this.tests.filter(t => evalEntryOutcome(t) === 'passed').length;
const manual = this.tests.filter(t => evalEntryOutcome(t) === 'manual-review').length;
const flaky = this.flakyRetries();
const result: EvalResult = {
@@ -948,8 +981,9 @@ export class EvalCollector {
total_tests: this.tests.length,
executed_tests: this.tests.filter(t => t.execution !== 'reused').length,
reused_tests: this.tests.filter(t => t.execution === 'reused').length,
...(manual ? { manual_accepted_tests: manual } : {}),
passed,
failed: this.tests.length - passed,
failed: this.tests.length - passed - manual,
total_cost_usd: Math.round(totalCost * 100) / 100,
total_duration_ms: totalDuration,
wall_clock_ms: Date.now() - this.createdAt,
@@ -999,7 +1033,9 @@ export class EvalCollector {
lines.push('═'.repeat(70));
for (const t of this.tests) {
const status = !t.passed ? ' FAIL ' : t.execution === 'reused' ? ' REUSE' : ' PASS ';
const outcome = evalEntryOutcome(t);
const status = outcome === 'manual-review' ? 'MANUAL' : outcome === 'failed' ? ' FAIL '
: t.execution === 'reused' ? ' REUSE' : ' PASS ';
const cost = `$${t.cost_usd.toFixed(2)}`;
const dur = t.duration_ms ? `${Math.round(t.duration_ms / 1000)}s` : '';
const turns = t.turns_used !== undefined ? `${t.turns_used}t` : '';
@@ -1010,6 +1046,8 @@ export class EvalCollector {
} else if (t.judge_scores) {
const scores = Object.entries(t.judge_scores).map(([k, v]) => `${k[0]}:${v}`).join(' ');
detail = scores;
} else if (outcome === 'manual-review') {
detail = `unscored; approved by ${t.manual_review!.approval.approved_by} (${t.manual_review!.approval.approval_url})`;
}
const name = t.name.length > 35 ? t.name.slice(0, 32) + '...' : t.name.padEnd(35);
@@ -1020,6 +1058,7 @@ export class EvalCollector {
const totalCost = `$${result.total_cost_usd.toFixed(2)}`;
const totalDur = `${Math.round(result.total_duration_ms / 1000)}s`;
lines.push(` Total: ${result.passed}/${result.total_tests} passed${' '.repeat(20)}${totalCost.padStart(6)} ${totalDur}`);
if (result.manual_accepted_tests) lines.push(` Manual accepted: ${result.manual_accepted_tests} unscored provider refusal(s)`);
lines.push(` Evidence: ${result.executed_tests ?? result.total_tests} executed, ${result.reused_tests ?? 0} reused`);
if (result.flaky_retries && result.flaky_retries.length > 0) {
// Loud, never fatal: a flaky pass must not block anyone, but it must
+33 -2
View File
@@ -13,7 +13,8 @@ import Anthropic from '@anthropic-ai/sdk';
import type { JSONOutputFormat } from '@anthropic-ai/sdk/resources/messages';
import { setTimeout as delay } from 'node:timers/promises';
import { CLAUDE_FRONTIER_EVAL_MODEL, resolveEvalModel } from '../../lib/eval-model';
import { CLAUDE_FRONTIER_EVAL_MODEL, DEFAULT_JUDGE_MAX_TOKENS, resolveEvalModel } from '../../lib/eval-model';
export { DEFAULT_JUDGE_MAX_TOKENS } from '../../lib/eval-model';
export interface JudgeScore {
clarity: number; // 1-5
@@ -22,6 +23,35 @@ export interface JudgeScore {
reasoning: string;
}
export interface JudgeRefusalEvidence {
stop_reason: 'refusal';
response_id: string | null;
request_id: string | null;
model: string | null;
input_tokens: number | null;
output_tokens: number | null;
text_blocks: number;
}
export class JudgeRefusalError extends Error {
readonly refusal: JudgeRefusalEvidence;
constructor(response: { id?: unknown; _request_id?: unknown; model?: unknown;
usage?: { input_tokens?: unknown; output_tokens?: unknown }; content: Array<{ type: string }> }) {
super('Judge provider refused the evaluation; no automated score');
this.name = 'JudgeRefusalError';
this.refusal = {
stop_reason: 'refusal',
response_id: typeof response.id === 'string' ? response.id : null,
request_id: typeof response._request_id === 'string' ? response._request_id : null,
model: typeof response.model === 'string' ? response.model : null,
input_tokens: typeof response.usage?.input_tokens === 'number' ? response.usage.input_tokens : null,
output_tokens: typeof response.usage?.output_tokens === 'number' ? response.usage.output_tokens : null,
text_blocks: response.content.filter(block => block.type === 'text').length,
};
}
}
export interface OutcomeJudgeResult {
detected: string[];
missed: string[];
@@ -89,7 +119,7 @@ export async function callJudge<T>(
// Thinking and answer text share max_tokens. The old 1024-token budget
// could be exhausted before a frontier judge emitted any JSON.
const resolvedModel = resolveEvalModel('judge', model);
const maxTokens = opts?.max_tokens ?? 8192;
const maxTokens = opts?.max_tokens ?? DEFAULT_JUDGE_MAX_TOKENS;
const client = new Anthropic();
const makeRequest = () => client.messages.create({
@@ -134,6 +164,7 @@ export async function callJudge<T>(
.map(block => block.text)
.join('\n');
try {
if (response.stop_reason === 'refusal') throw new JudgeRefusalError(response);
if (opts?.jsonSchema !== undefined) {
if (response.stop_reason !== 'end_turn') throw new Error(`Structured judge did not complete: stop_reason=${response.stop_reason}`);
return JSON.parse(text) as T;
@@ -0,0 +1,17 @@
import { readFileSync } from 'node:fs';
import { resolve } from 'node:path';
import type { EvalTestEntry } from './eval-store';
import { buildCookieWorkflowJudgeInput } from './cookie-workflow-judge-input';
import { COOKIE_MANUAL_REVIEW_FILE } from './cookie-workflow-manual-review';
export function manualReviewFixture(root = resolve(import.meta.dir, '../..')): EvalTestEntry {
const approval = JSON.parse(readFileSync(resolve(root, COOKIE_MANUAL_REVIEW_FILE), 'utf8'));
return {
name: 'setup-browser-cookies/SKILL.md workflow', suite: 'Cookie setup workflow quality', tier: 'llm-judge',
passed: false, execution: 'executed', exit_reason: 'provider_refusal', attempt: 1, duration_ms: 1, cost_usd: 0,
model: approval.model, prompt: buildCookieWorkflowJudgeInput(root).prompt,
error: 'Synthetic provider refusal fixture, not live model evidence',
manual_review: { approval, refusal: { stop_reason: 'refusal', response_id: 'msg_synthetic_fixture',
request_id: 'req_synthetic_fixture', model: approval.model, input_tokens: 1, output_tokens: 0, text_blocks: 0 } },
};
}
+38 -11
View File
@@ -830,17 +830,31 @@ export type SharedQuestionSelector = (input: Record<string, unknown>) => Record<
/** The skip actor may decline work, never approve a mixed fix/preservation choice. */
function skippedReviewOption(question: any): any {
const options = Array.isArray(question?.options) ? question.options : [];
const qualifiedIndexState = /^(?:(?:git\s+)?index|skip-worktree|assume-unchanged)\s+(?:bits?|flags?|attributes?|settings?)$/i;
const candidates = options.flatMap((option: any) => {
if (typeof option?.label !== 'string' || ['description', 'preview'].some(field =>
option[field] !== undefined && typeof option[field] !== 'string')) return [];
const label = option.label.replace(/[‘’]/g, "'").replace(/^\s*(?:[A-Z]|\d+)[.)]\s*/i, '')
.replace(/\s*\(recommended\)\s*$/i, '').trim();
const label = option.label.replace(/[‘’]/g, "'").replace(/`/g, '').replace(/^\s*(?:[A-Z]|\d+)[.)]\s*/i, '')
.replace(/\s*\(recommended\)\s*$/i, '').trim()
.replace(/^no\s*[,.:!?]\s*(?=(?:skip|decline|keep|leave|do not|don't)\b)/i, '');
const referentialRetention = /^(?:keep|leave)\s+(?:it|this|that|them|these)$/i.test(label);
const preservation = option.description?.trim().replace(/`/g, '').match(/^(?:keep|leave|retain|preserve)\s+([^,;.!?]+)/i);
const preservedObject = preservation?.[1].split(/\b(?:and|but|while)\b/i)[0]
.replace(/(?<![-\w])(?:the|this|that|current|existing|local|as[- ]is|unchanged|untouched|set|hidden)\b/gi, '').trim();
const describedRetention = !!preservedObject
&& !/^\w+ing\b/i.test(preservedObject)
&& (/^(?:(?:duplicated|original|prior|tracked|untracked)\s+)*(?:(?:index|skip-worktree|assume-unchanged)\s+)?(?:flags?|code|source|implementations?|copies|copy|files?|routes?|workers?|helpers?|parsers?|changes?|contents?|state|branches|branch|worktrees?)$/i.test(preservedObject)
|| qualifiedIndexState.test(preservedObject));
const description = (option.description ?? '').replace(/[‘’]/g, "'").trim();
const declinesChange = /^(?:do not|don't)\s+(?:apply|change|edit|fix|refactor|extract|modify|touch|clear|remove|update|replace|add|migrate|implement|reuse|import)\b/i;
const labelObject = label.match(/^(?:keep|leave)\s+(?:the\s+)?(.+)$/i)?.[1]
.replace(/\s+(?:as[- ]is|unchanged|untouched|set)$/i, '');
const preservationRank = referentialRetention ? describedRetention || declinesChange.test(description)
: /^(?:keep|leave)\b.*\b(?:current|existing|unchanged|untouched|as[- ]is|alone|set|copies|copy|implementation|code|source)\b/i.test(label)
|| !!labelObject && qualifiedIndexState.test(labelObject);
const rank = /^(?:skip|decline)(?=$|\s|[,.!])/i.test(label) ? 3
: declinesChange.test(label) ? 2
: /^(?:keep|leave)\b.*\b(?:current|existing|unchanged|untouched|as[- ]is|alone|set|copies|copy|implementation|code|source)\b/i.test(label)
|| (/^(?:keep|leave)\b/i.test(label) && declinesChange.test(description)) ? 1 : 0;
: preservationRank ? 1 : 0;
if (!rank) return [];
// A leading decline names rejected work. Classify later commitments rather
// than action words inside recorded metadata or hypothetical consequences.
@@ -848,21 +862,34 @@ function skippedReviewOption(question: any): any {
option.description ?? '', option.preview ?? ''].join('\n').replace(/[‘’]/g, "'");
const actions = new Set(['approve', 'fix', 'apply', 'refactor', 'extract', 'replace', 'rewrite', 'edit', 'modify',
'change', 'clear', 'remove', 'delete', 'add', 'update', 'implement', 'migrate', 'touch', 're-export',
'import', 'reuse', 'share', 'wire', 'convert']);
'import', 'reuse', 'share', 'wire', 'convert', 'set', 'unset', 'toggle', 'flip', 'reset', 'enable', 'disable']);
const isAction = (word = '') => [word, word.replace(/s$/, ''), word.replace(/(?:es|ed|ing)$/, ''),
word.replace(/(?:ed|ing)$/, 'e'), word.replace(/(?:ies|ied)$/, 'y')].some(form => actions.has(form));
word.replace(/(?:ed|ing)$/, 'e'), word.replace(/(?:ies|ied)$/, 'y'),
word.replace(/([a-z])\1(?:ed|ing)$/, '$1')].some(form => actions.has(form));
const changes = commitment.toLowerCase().split(/[,;\n]|[.!?](?:\s|$)|\b(?:and|but|then|while)\b/).some(part => {
const clause = part.replace(/^[^a-z]+/, '')
.replace(/^(?:(?:this|that|the|selected|chosen)\s+(?:option|choice|selection)|i|we|you|it|(?:the\s+)?(?:source|code|route|worker|helper|parser|index(?:\s+flag)?))\s+/, '')
.replace(/^(?:will|would|should|must|can|may|does|do)\s+/, '')
.replace(/^(?:(?:please|also|still|just|now|be)\s+)+/, '');
if (/^(?:not|does not|don't|doesn't|won't|without|no)\b/.test(clause)) return false;
// The no-change choice may persist/reuse its review decision. That is not
// permission to modify source or clear an index flag.
if (/^(?:updates?|updated|updating|reuses?|reused|reusing)\s+(?:the\s+)?(?:(?:prior|recorded|existing)\s+)?(?:review\s+(?:log|record)|decision|advisory|snapshot|ledger)\b/.test(clause)) return false;
const first = clause.match(/^[a-z]+(?:-[a-z]+)*/)?.[0];
const future = clause.match(/\bwill\s+(?:be\s+)?([a-z]+(?:-[a-z]+)*)/)?.[1];
return isAction(first) || isAction(future);
const futureMatch = clause.match(/\b(?:will|would|should|must|can|may)\s+(?:(?:still|also|now|just|[a-z]+ly)\s+)*(?:be\s+)?(?:(?:still|also|now|just|[a-z]+ly)\s+)*([a-z]+(?:-[a-z]+)*)/);
const future = futureMatch?.[1];
const method = /^(?:keep|leave|retain|preserve)\b/.test(clause)
&& [...clause.matchAll(/\b(?:by|through|via)\s+([a-z]+(?:-[a-z]+)*)/g)].some(match => isAction(match[1]));
const state = /^(?:change|modification|file|flag|state|content)\s+(?:stays?|remains?)\b/.test(clause);
const recordedDecision = /^(?:updates?|updated|updating|reuses?|reused|reusing)\s+(?:the\s+)?(?:(?:prior|recorded|existing)\s+)?(?:review\s+(?:log|record)|decision|advisory|snapshot|ledger)\b/.test(clause);
const nominalReuse = /^(?:the\s+)?reuse\s+of\b/.test(clause);
const describedReuse = /\b(?:is|are|was|were|remains?|stays?|requires?|needs?|will|would|should|must|can|may)\s+(?:(?:still|also|now|just|not|never|[a-z]+ly)\s+)*[a-z]+(?:-[a-z]+)*/.test(clause);
const futureSubject = clause.slice(0, futureMatch?.index ?? 0).trim();
const passiveDecision = /\b(?:review\s+(?:log|record)|decision|advisory|snapshot|ledger)$/.test(futureSubject)
|| /\b(?:review\s+(?:log|record)|decision|advisory|snapshot|ledger)\b(?:(?!\b(?:source|code|route|worker|helper|parser|file|flag)\b).)*\bit$/.test(futureSubject);
const futureDecision = /\b(?:can|will|would|should|must|may)\s+(?:(?:still|also|now|just|[a-z]+ly)\s+)*reuse\s+(?:(?:this|the|prior|recorded|existing)\s+)*(?:review\s+(?:log|record)|decision|advisory|snapshot|ledger)\b/.test(clause);
const purpose = [...clause.matchAll(/\b(?:to|by|through|via)\s+(?:[a-z]+ly\s+)*([a-z]+(?:-[a-z]+)*)/g)]
.some(match => isAction(match[1]));
return (isAction(future) && !(future === 'reused' && passiveDecision) && !(future === 'reuse' && futureDecision)) || method || purpose
|| (nominalReuse && !describedReuse)
|| (!state && !recordedDecision && !nominalReuse && isAction(first));
});
return changes ? [] : [{ option, rank }];
});
+1
View File
@@ -1772,6 +1772,7 @@ export const E2E_TIERS: Record<string, 'gate' | 'periodic'> = {
* LLM-judge test touchfiles — keyed by test description string.
*/
export const LLM_JUDGE_TOUCHFILES: Record<string, string[]> = {
'setup-browser-cookies/SKILL.md workflow': ['setup-browser-cookies/SKILL.md.tmpl', 'setup-browser-cookies/SKILL.md', 'BROWSER.md', 'test/helpers/cookie-workflow-judge-input.ts', 'test/cookie-workflow-judge-input.test.ts', 'test/helpers/cookie-workflow-manual-review.ts', 'test/cookie-workflow-manual-review.test.ts', 'test/helpers/manual-judge-review-fixture.ts', '.github/cookie-workflow-manual-review.json', 'test/helpers/workflow-judge-input.ts', 'test/skill-llm-eval.test.ts'],
'command reference table': ['browse/sections/**', 'SKILL.md', 'SKILL.md.tmpl', 'browse/src/commands.ts', 'test/skill-llm-eval.test.ts'],
'snapshot flags reference': ['browse/sections/**', 'SKILL.md', 'SKILL.md.tmpl', 'browse/src/snapshot.ts', 'test/skill-llm-eval.test.ts'],
'browse/SKILL.md reference': ['browse/sections/**', 'browse/SKILL.md', 'browse/SKILL.md.tmpl', 'browse/src/**', 'test/skill-llm-eval.test.ts'],
+2 -2
View File
@@ -3,7 +3,7 @@ import * as fs from 'node:fs';
import * as path from 'node:path';
import { isBuiltin } from 'node:module';
import { spawnSync } from 'node:child_process';
import { resolveEvalModel } from '../../lib/eval-model';
import { DEFAULT_JUDGE_MAX_TOKENS, resolveEvalModel } from '../../lib/eval-model';
import { JUDGE_MS } from './eval-budgets';
import type { JudgeScore } from './llm-judge';
import { readWorkflowJudgeInput, buildWorkflowJudgePrompt } from './workflow-judge-input';
@@ -98,7 +98,7 @@ export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): {
coverage: { dependencies: 'complete', prompts: 'complete', environment: 'complete' }, unknownDependencies: [],
files: workflowJudgeDependencies(opts.root, input.files.map(file => file.path)),
prompts: { [opts.testName]: prompt },
parameters: { rootPackage, thresholds: opts.thresholds, max_tokens: 8192, temperature: null, budget_ms: JUDGE_MS,
parameters: { rootPackage, thresholds: opts.thresholds, max_tokens: DEFAULT_JUDGE_MAX_TOKENS, temperature: null, budget_ms: JUDGE_MS,
request: 'messages.create/user', retries: 1 },
runtime: { image: env.EVALS_CACHE_RUNTIME_ID!, bun: Bun.version, node: process.versions.node,
platform: process.platform, arch: process.arch, judge: resolveEvalModel('judge', undefined, env),