mirror of
https://github.com/garrytan/gstack.git
synced 2026-09-28 07:32:14 +02:00
v1.90.0.0 feat: make browser cookie imports explicit and safe (#2964)
* fix(browse): prepare reliable cookie import wave for validation * ci: sequence quality and behavior for validation branch * fix(browse): isolate Windows qualification and preserve native diagnostics * test(browse): cover cookie workflow quality and isolate Windows user paths * test(browse): trace native member startup and initialize fresh folders * fix(browse): keep Windows member stdin alive through EOF * fix(browse): latch native timeouts and compare contained Edge startup * test(browse): verify native version metadata and actual Windows argv * test(browse): qualify Dia import on isolated macOS CI * fix(browse): require picker origin for session mutations * fix(browse): bound credential reads through stream completion * test(browse): inspect owned Windows process arguments natively * test(evals): preserve passing coverage during cookie repair reruns * test(browse): isolate Dia qualification in a fresh macOS account * test(browse): pass bounded integer timeouts to native Mac probes * test(browse): distinguish Windows profile initialization from containment * test(browse): await descendant pipe readiness before parent exit * test(browse): initialize and restore isolated macOS Keychain state * test(browse): initialize Windows fixture folders before qualification * test(ci): pin the same Node runtime across Windows checks * test(browse): distinguish native macOS browser preflight stages * test(browse): isolate Windows descendant console lifetime * test(browse): preserve native receipts and identify fixture lock holders * test(browse): prepare dependency resolution before native Mac worker startup * test(ci): include lock and close checks in native diagnostics * test(browse): preserve native owner probe stages and subprocess deadlines * fix(browse): classify Chromium profile-in-use exit precisely * test(browse): retain Mac qualification evidence through cleanup failures * test(browse): bound Mac fixture paths and retire its owned user domain * test(browse): accept vanished fixture entries without weakening cleanup * test(browse): identify probe-created macOS user domains safely * test(browse): observe Mac user domains without targeting them first * test(browse): use passive fresh-user ownership throughout Mac qualification * test(browse): distinguish profile and registered-home Keychain lookups * test(browse): qualify Dia under one registered account home * test(browse): identify Dia startup and owned process-group failures * test(browse): classify bounded Dia startup diagnostics without leaking output * fix(test): preserve native Mac sandboxing and reap owned browser children * fix(browse): preserve Chromium sandboxing for native profile imports * test(browse): inspect signed Mach-O architecture without launching Xcode tools * test(browse): sample pending Dia startup and reap on all cleanup paths * test(browse): compare protected Dia launches in fresh Bun and Node accounts * test(browse): inspect isolated Mac GUI readiness without browser access * v1.90.0.0 fix: bind cookie picker actions to their document * test: validate cookie guards and fit nested launch fixtures * ci: configure the bundled Chromium sandbox helper * fix(browse): classify Playwright authentication timeouts * test: retain bounded Windows lifecycle diagnostics * test(cso): reuse bounded NTFS precision candidates * test(review): handle explicit preservation choices safely * test(browse): remove owned fixture directories with explicit primitives * test(review): distinguish descriptive reuse from edit commitments * test: admit only the approved unscored cookie workflow refusal * test: keep the Office Hours judge mock export-complete * fix: keep dependency-free CI planners independent of the model SDK * test: observe the exact holder after a native fixture unlink failure * fix: start seeded PTY observations at owned readiness * test: acquire identity-bound Windows deletion admission before profile resets * test: preserve qualified Git index bits without authorizing mutations
This commit is contained in:
@@ -4605,7 +4605,7 @@ export async function runPlanSkillObservation(opts: {
|
||||
};
|
||||
// Entry deadline → boot → owned paste/receipt/ack → slash → observation.
|
||||
// Setup consumes the existing case budget; cleanup has its separate grace.
|
||||
await Bun.sleep(Math.min(8000, Math.max(0, deadlineAt - Date.now())));
|
||||
if (!opts.initialPlanContent) await Bun.sleep(Math.min(8000, Math.max(0, deadlineAt - Date.now())));
|
||||
if (opts.initialPlanContent) {
|
||||
const seed = `Keep this draft plan as context. Briefly acknowledge receipt, then wait for my next message containing a slash command. Do not start the review or call tools yet.\n\n${opts.initialPlanContent}`;
|
||||
try {
|
||||
|
||||
@@ -0,0 +1,50 @@
|
||||
import { createHash } from 'node:crypto';
|
||||
import { readFileSync } from 'node:fs';
|
||||
import { join } from 'node:path';
|
||||
import { buildWorkflowJudgePrompt, type WorkflowJudgeFile, type WorkflowJudgeInput } from './workflow-judge-input';
|
||||
|
||||
export const COOKIE_WORKFLOW_JUDGE = {
|
||||
judgeContext: 'a fallback-browser cookie import workflow',
|
||||
judgeGoal: 'how to select an authorized source browser, profile, and domain without guessing an account; configure optional authentication verification before mutation; obtain explicit consent for precisely scoped storage reset; distinguish copied cookies from positive sign-in evidence; and recover within the documented platform and privacy boundaries',
|
||||
thresholds: { clarity: 4, completeness: 3, actionability: 4 },
|
||||
} as const;
|
||||
|
||||
export function buildCookieWorkflowJudgeInput(root: string): WorkflowJudgeInput & { prompt: string; sha256: string } {
|
||||
const files: WorkflowJudgeFile[] = [
|
||||
{ path: 'setup-browser-cookies/SKILL.md', kind: 'entrypoint', start: '# Setup Browser Cookies', end: null },
|
||||
{ path: 'BROWSER.md', kind: 'section', start: '#### Choosing a source and checking sign-in', end: '### Tabs + frames' },
|
||||
].map(spec => {
|
||||
const source = readFileSync(join(root, spec.path), 'utf8');
|
||||
const locate = (marker: string): number => {
|
||||
const matches = [...source.matchAll(new RegExp(`^${marker.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}\\r?$`, 'gm'))];
|
||||
if (matches.length !== 1) throw new Error(`${spec.path}: expected exactly one marker ${JSON.stringify(marker)}, found ${matches.length}`);
|
||||
return matches[0].index!;
|
||||
};
|
||||
const start = locate(spec.start);
|
||||
const end = spec.end === null ? source.length : locate(spec.end);
|
||||
if (end <= start || !source.slice(start + spec.start.length, end).trim()) {
|
||||
throw new Error(`${spec.path}: empty or reversed cookie workflow excerpt`);
|
||||
}
|
||||
return {
|
||||
path: spec.path,
|
||||
kind: spec.kind as WorkflowJudgeFile['kind'],
|
||||
content: source.slice(start, end),
|
||||
startLine: source.slice(0, start).split('\n').length,
|
||||
endLine: source.slice(0, end - 1).split('\n').length,
|
||||
};
|
||||
});
|
||||
if (!files[0].content.includes('`BROWSER.md`') || !files[0].content.includes('**Choosing a source and checking sign-in**')) {
|
||||
throw new Error('setup-browser-cookies/SKILL.md: missing cookie reference link');
|
||||
}
|
||||
const text = [
|
||||
'SKILL.md is the entry point. BROWSER.md supplies the exact referenced cookie section, not an additional execution step.',
|
||||
...files.map(file => [
|
||||
`--- BEGIN FILE ${JSON.stringify(file.path)} (lines ${file.startLine}-${file.endLine}; ${file.kind}) ---`,
|
||||
file.content,
|
||||
`--- END FILE ${JSON.stringify(file.path)} ---`,
|
||||
].join('\n')),
|
||||
].join('\n\n');
|
||||
const input = { files, text };
|
||||
const prompt = buildWorkflowJudgePrompt(COOKIE_WORKFLOW_JUDGE, input);
|
||||
return { ...input, prompt, sha256: createHash('sha256').update(prompt).digest('hex') };
|
||||
}
|
||||
@@ -0,0 +1,101 @@
|
||||
import { createHash } from 'node:crypto';
|
||||
import { readFileSync } from 'node:fs';
|
||||
import { join } from 'node:path';
|
||||
import { buildCookieWorkflowJudgeInput, COOKIE_WORKFLOW_JUDGE } from './cookie-workflow-judge-input';
|
||||
import type { JudgeRefusalEvidence } from './llm-judge';
|
||||
import { DEFAULT_JUDGE_MAX_TOKENS, resolveEvalModel } from '../../lib/eval-model';
|
||||
|
||||
export const COOKIE_MANUAL_REVIEW_FILE = '.github/cookie-workflow-manual-review.json';
|
||||
const CASE = 'setup-browser-cookies/SKILL.md workflow';
|
||||
type Thresholds = { clarity: number; completeness: number; actionability: number };
|
||||
|
||||
export interface CookieManualApproval {
|
||||
schema_version: 1;
|
||||
test_name: typeof CASE;
|
||||
prompt_sha256: string;
|
||||
prompt_bytes: number;
|
||||
model: string;
|
||||
max_tokens: number;
|
||||
thresholds: Thresholds;
|
||||
approved_by: string;
|
||||
approved_at: string;
|
||||
approval_url: string;
|
||||
reason: string;
|
||||
}
|
||||
|
||||
export interface ManualJudgeReview {
|
||||
approval: CookieManualApproval;
|
||||
refusal: JudgeRefusalEvidence;
|
||||
}
|
||||
|
||||
const object = (value: unknown): value is Record<string, any> => value !== null && typeof value === 'object' && !Array.isArray(value);
|
||||
const nonempty = (value: unknown): value is string => typeof value === 'string' && value.trim().length > 0;
|
||||
const dimensions = ['clarity', 'completeness', 'actionability'] as const;
|
||||
const sameThresholds = (a: Thresholds, b: Thresholds) => dimensions.every(key => a[key] === b[key]);
|
||||
|
||||
function validApproval(value: unknown): value is CookieManualApproval {
|
||||
return object(value) && value.schema_version === 1 && value.test_name === CASE
|
||||
&& typeof value.prompt_sha256 === 'string' && /^[a-f0-9]{64}$/.test(value.prompt_sha256)
|
||||
&& Number.isSafeInteger(value.prompt_bytes) && value.prompt_bytes > 0
|
||||
&& nonempty(value.model) && Number.isSafeInteger(value.max_tokens) && value.max_tokens > 0
|
||||
&& object(value.thresholds) && dimensions.every(key => Number.isInteger(value.thresholds[key]) && value.thresholds[key] >= 1 && value.thresholds[key] <= 5)
|
||||
&& nonempty(value.approved_by) && typeof value.approved_at === 'string' && /^\d{4}-\d{2}-\d{2}$/.test(value.approved_at)
|
||||
&& typeof value.approval_url === 'string' && /^https:\/\/github\.com\/garrytan\/gstack\/pull\/2964#issuecomment-\d+$/.test(value.approval_url)
|
||||
&& nonempty(value.reason);
|
||||
}
|
||||
|
||||
function validRefusal(value: unknown, model: string): value is JudgeRefusalEvidence {
|
||||
return object(value) && value.stop_reason === 'refusal' && value.model === model
|
||||
&& nonempty(value.response_id) && nonempty(value.request_id)
|
||||
&& Number.isSafeInteger(value.input_tokens) && value.input_tokens >= 0
|
||||
&& value.output_tokens === 0 && value.text_blocks === 0;
|
||||
}
|
||||
|
||||
export function isManualReviewEntry(entry: unknown): boolean {
|
||||
if (!object(entry) || entry.name !== CASE || entry.suite !== 'Cookie setup workflow quality'
|
||||
|| entry.tier !== 'llm-judge' || entry.passed !== false
|
||||
|| entry.attempt !== 1
|
||||
|| entry.execution !== 'executed' || entry.exit_reason !== 'provider_refusal'
|
||||
|| entry.judge_scores !== undefined || entry.judge_reasoning !== undefined || entry.reused_from !== undefined
|
||||
|| typeof entry.prompt !== 'string' || !object(entry.manual_review)) return false;
|
||||
const { approval, refusal } = entry.manual_review;
|
||||
return validApproval(approval) && entry.model === approval.model && validRefusal(refusal, approval.model)
|
||||
&& Buffer.byteLength(entry.prompt) === approval.prompt_bytes
|
||||
&& createHash('sha256').update(entry.prompt).digest('hex') === approval.prompt_sha256;
|
||||
}
|
||||
|
||||
export function getCookieWorkflowManualReview(root: string, request: {
|
||||
testName: string; prompt: string; model: string; maxTokens: number; thresholds: Thresholds; attempt: number;
|
||||
}, refusal: JudgeRefusalEvidence): ManualJudgeReview | null {
|
||||
if (request.testName !== CASE || request.attempt !== 1 || !validRefusal(refusal, request.model)) return null;
|
||||
let approval: unknown;
|
||||
try { approval = JSON.parse(readFileSync(join(root, COOKIE_MANUAL_REVIEW_FILE), 'utf8')); }
|
||||
catch (error) {
|
||||
if ((error as NodeJS.ErrnoException).code === 'ENOENT') return null;
|
||||
throw error;
|
||||
}
|
||||
if (!validApproval(approval)) throw new Error('Invalid cookie workflow manual-review approval');
|
||||
const current = buildCookieWorkflowJudgeInput(root);
|
||||
if (request.prompt !== current.prompt || current.sha256 !== approval.prompt_sha256
|
||||
|| Buffer.byteLength(request.prompt) !== approval.prompt_bytes || request.model !== approval.model
|
||||
|| request.maxTokens !== approval.max_tokens || request.maxTokens !== DEFAULT_JUDGE_MAX_TOKENS
|
||||
|| !sameThresholds(request.thresholds, approval.thresholds)
|
||||
|| !sameThresholds(request.thresholds, COOKIE_WORKFLOW_JUDGE.thresholds)) return null;
|
||||
return { approval, refusal };
|
||||
}
|
||||
|
||||
export function manualReviewProblem(entry: unknown, root: string): string | null {
|
||||
if (!object(entry) || !Object.hasOwn(entry, 'manual_review')) return null;
|
||||
if (!isManualReviewEntry(entry)) return 'Malformed manual-review claim';
|
||||
if (entry.model !== resolveEvalModel('judge')) return 'Manual review model is not the current judge model';
|
||||
try {
|
||||
const claimed = entry.manual_review as ManualJudgeReview;
|
||||
const verified = getCookieWorkflowManualReview(root, { testName: entry.name, prompt: entry.prompt,
|
||||
model: entry.model, maxTokens: claimed.approval.max_tokens, thresholds: claimed.approval.thresholds,
|
||||
attempt: entry.attempt }, claimed.refusal);
|
||||
if (!verified || Object.entries(verified.approval).some(([key, value]) => key === 'thresholds'
|
||||
? !sameThresholds(value as Thresholds, claimed.approval.thresholds)
|
||||
: value !== claimed.approval[key as keyof CookieManualApproval])) return 'Manual review does not match current source and approval';
|
||||
return null;
|
||||
} catch { return 'Manual review approval or current input is unavailable or invalid'; }
|
||||
}
|
||||
@@ -0,0 +1,34 @@
|
||||
import * as fs from 'node:fs';
|
||||
import * as path from 'node:path';
|
||||
|
||||
export function createPrecisionLossCandidate(
|
||||
target: string,
|
||||
contents: string,
|
||||
readFileId: (file: string) => bigint = file => fs.lstatSync(file, { bigint: true }).ino,
|
||||
): bigint {
|
||||
if (fs.existsSync(target)) throw new Error('Precision fixture target already exists');
|
||||
const directory = path.dirname(target);
|
||||
if (fs.lstatSync(directory).isSymbolicLink()) throw new Error('Precision fixture directory must not be linked');
|
||||
const pool = fs.mkdtempSync(path.join(directory, 'ntfs-id-'));
|
||||
let aboveSafe = 0;
|
||||
try {
|
||||
for (let batch = 0; batch < 512; batch++) {
|
||||
const files: string[] = [];
|
||||
for (let slot = 0; slot < 2; slot++) {
|
||||
const file = path.join(pool, `candidate-${slot}`);
|
||||
fs.writeFileSync(file, contents, { flag: 'wx' });
|
||||
files.push(file);
|
||||
const inode = readFileId(file);
|
||||
if (inode > BigInt(Number.MAX_SAFE_INTEGER)) aboveSafe++;
|
||||
if (inode > BigInt(Number.MAX_SAFE_INTEGER) && String(Number(inode)) !== String(inode)) {
|
||||
fs.linkSync(file, target);
|
||||
return inode;
|
||||
}
|
||||
}
|
||||
for (const file of files) fs.unlinkSync(file);
|
||||
}
|
||||
throw new Error(`No precision-losing NTFS file ID within 1024 file creations (${aboveSafe} above-safe observations)`);
|
||||
} finally {
|
||||
fs.rmSync(pool, { recursive: true, force: true });
|
||||
}
|
||||
}
|
||||
@@ -107,7 +107,7 @@ export const FILE_RETRY_BUDGETS = [
|
||||
...[
|
||||
// Fourteen workflow judges include their 10s recording grace; the other
|
||||
// eleven judges retain 120s. Supervise all 25 and the existing one retry.
|
||||
{ file: 'test/skill-llm-eval.test.ts', attemptMs: 14 * (JUDGE_MS + 10_000) + 11 * JUDGE_MS, retries: 1 },
|
||||
{ file: 'test/skill-llm-eval.test.ts', attemptMs: 15 * (JUDGE_MS + 10_000) + 11 * JUDGE_MS, retries: 1 },
|
||||
{ file: 'test/codex-e2e-plan-format.test.ts', attemptMs: 4 * (CAPTURE_LONG_MS + 10_000), retries: 1 },
|
||||
{ file: 'test/skill-e2e-auq-matrix.test.ts', attemptMs: 6 * CAPTURE_MS, retries: 1 },
|
||||
{ file: 'test/skill-e2e-plan-format.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), retries: 1 },
|
||||
|
||||
@@ -10,11 +10,13 @@ import {
|
||||
isPartialEval,
|
||||
listEvalJsonFiles,
|
||||
compareEvalResults,
|
||||
evalEntryOutcome,
|
||||
formatComparison,
|
||||
generateCommentary,
|
||||
judgePassed,
|
||||
} from './eval-store';
|
||||
import type { EvalResult, EvalTestEntry, ComparisonResult } from './eval-store';
|
||||
import { manualReviewFixture } from './manual-judge-review-fixture';
|
||||
|
||||
let tmpDir: string;
|
||||
|
||||
@@ -77,6 +79,36 @@ async function captureStderr(fn: () => Promise<void>): Promise<string> {
|
||||
// --- EvalCollector tests ---
|
||||
|
||||
describe('EvalCollector', () => {
|
||||
test('manual provider refusal stays unscored and executed; malformed claims stay failed', async () => {
|
||||
const manual = manualReviewFixture();
|
||||
const invalidPass = { ...manual, passed: true };
|
||||
const invalidScore = { ...manual, judge_scores: { clarity: 5 } };
|
||||
expect(evalEntryOutcome(manual)).toBe('manual-review');
|
||||
expect(evalEntryOutcome(invalidPass)).toBe('failed');
|
||||
expect(evalEntryOutcome(invalidScore)).toBe('failed');
|
||||
expect(evalEntryOutcome({ ...manual, manual_review: { ...manual.manual_review, approval: { ...manual.manual_review!.approval,
|
||||
prompt_sha256: '0'.repeat(64) } } })).toBe('failed');
|
||||
const collector = new EvalCollector('llm-judge', tmpDir);
|
||||
collector.addTest(makeEntry({ name: 'ordinary', tier: 'llm-judge' }));
|
||||
collector.addTest(manual);
|
||||
collector.addTest(invalidPass);
|
||||
collector.addTest(invalidScore);
|
||||
const partial: EvalResult = JSON.parse(fs.readFileSync(path.join(tmpDir, '_partial-e2e.json'), 'utf8'));
|
||||
expect(partial).toMatchObject({ passed: 1, failed: 2, manual_accepted_tests: 1,
|
||||
executed_tests: 4, reused_tests: 0 });
|
||||
const output = await captureStderr(async () => { await collector.finalize(); });
|
||||
const result: EvalResult = JSON.parse(fs.readFileSync(fs.readdirSync(tmpDir).map(name => path.join(tmpDir, name))
|
||||
.find(name => name.endsWith('.json') && !path.basename(name).startsWith('_'))!, 'utf8'));
|
||||
expect(result).toMatchObject({ passed: 1, failed: 2, manual_accepted_tests: 1, executed_tests: 4 });
|
||||
expect(result.tests[1]).toMatchObject({ passed: false, execution: 'executed', exit_reason: 'provider_refusal' });
|
||||
expect(result.tests[1].judge_scores).toBeUndefined();
|
||||
expect(output).toContain('MANUAL');
|
||||
expect(output).toContain('1 unscored provider refusal');
|
||||
expect(output).toContain(manual.manual_review!.approval.approval_url);
|
||||
expect(result.tests[2].passed).toBe(true);
|
||||
expect(output).toContain(' FAIL ');
|
||||
});
|
||||
|
||||
test('reused passing evidence preserves origin and remains separate from newly executed attempts', async () => {
|
||||
const collector = new EvalCollector('llm-judge', tmpDir);
|
||||
const reused_from = { input_key: 'a'.repeat(64), run_id: '1234/1', revision: 'b'.repeat(40),
|
||||
@@ -590,6 +622,45 @@ describe('findLatestFinalizedRun', () => {
|
||||
// --- compareEvalResults tests ---
|
||||
|
||||
describe('compareEvalResults', () => {
|
||||
test('manual acceptance is neither a score regression nor a recovery', () => {
|
||||
const manual = manualReviewFixture();
|
||||
const prior = makeResult({ tests: [makeEntry({ name: manual.name, passed: true })] });
|
||||
const accepted = makeResult({ tests: [manual], passed: 0, failed: 0, manual_accepted_tests: 1 });
|
||||
const toManual = compareEvalResults(prior, accepted, 'prior.json', 'accepted.json');
|
||||
expect(toManual).toMatchObject({ improved: 0, regressed: 0, manual_reviewed: 1 });
|
||||
expect(toManual.deltas[0]).toMatchObject({ status_change: 'manual-review', before: { passed: true },
|
||||
after: { passed: false, manual_review: true } });
|
||||
expect(formatComparison(toManual)).toContain('PASS → MANUAL');
|
||||
expect(formatComparison(toManual)).not.toContain('REGRESSION:');
|
||||
const fromManual = compareEvalResults(accepted, prior, 'accepted.json', 'prior.json');
|
||||
expect(fromManual).toMatchObject({ improved: 0, regressed: 0, manual_reviewed: 1 });
|
||||
expect(formatComparison(fromManual)).toContain('MANUAL → PASS');
|
||||
const invalid = makeResult({ tests: [{ ...manual, passed: true }] });
|
||||
expect(compareEvalResults(prior, invalid, 'prior.json', 'invalid.json').regressed).toBe(1);
|
||||
});
|
||||
|
||||
test('manual acceptance followed by a real or malformed failure is a blocking regression', () => {
|
||||
const manual = manualReviewFixture();
|
||||
const accepted = makeResult({ tests: [manual], passed: 0, failed: 0, manual_accepted_tests: 1 });
|
||||
const { manual_review: _receipt, ...ordinary } = manual;
|
||||
for (const after of [
|
||||
{ ...ordinary, exit_reason: 'timeout' },
|
||||
{ ...manual, passed: true },
|
||||
{ ...manual, judge_scores: { clarity: 5 } },
|
||||
]) {
|
||||
const failed = makeResult({ tests: [after] });
|
||||
const comparison = compareEvalResults(accepted, failed, 'accepted.json', 'failed.json');
|
||||
expect(comparison).toMatchObject({ improved: 0, regressed: 1 });
|
||||
expect(comparison.manual_reviewed).toBeUndefined();
|
||||
expect(comparison.deltas[0]).toMatchObject({ status_change: 'regressed', before: { manual_review: true },
|
||||
after: { passed: false } });
|
||||
const output = formatComparison(comparison);
|
||||
expect(output).toContain('MANUAL → FAIL');
|
||||
expect(output).toContain('REGRESSION:');
|
||||
expect(output).toContain('blocking failure');
|
||||
}
|
||||
});
|
||||
|
||||
test('detects improved/regressed/unchanged per test', () => {
|
||||
const before = makeResult({
|
||||
tests: [
|
||||
|
||||
+57
-18
@@ -12,6 +12,8 @@ import * as fs from 'fs';
|
||||
import * as path from 'path';
|
||||
import * as os from 'os';
|
||||
import { spawnSync } from 'child_process';
|
||||
import { isManualReviewEntry } from './cookie-workflow-manual-review';
|
||||
import type { ManualJudgeReview } from './cookie-workflow-manual-review';
|
||||
|
||||
// v2: EvalTestEntry.harvest gains optional {insertions, deletions, net} and
|
||||
// may be explicitly null (arm-benchmark harvest-failure taxonomy). Readers
|
||||
@@ -66,6 +68,7 @@ export interface EvalTestEntry {
|
||||
/** Absent in older records means executed; reuse is never a new model run. */
|
||||
execution?: 'executed' | 'reused';
|
||||
reused_from?: { input_key: string; run_id: string; revision: string; completed_at: string };
|
||||
manual_review?: ManualJudgeReview;
|
||||
/** 1-based record attempt for this name in this run. bun's --retry leaves
|
||||
* retried passes INVISIBLE in its text output (a fail→pass prints no
|
||||
* (fail) line and recaps as a clean pass — probed on 1.3.10), so the ONLY
|
||||
@@ -120,6 +123,14 @@ export interface EvalTestEntry {
|
||||
} | null;
|
||||
}
|
||||
|
||||
export function evalEntryOutcome(entry: unknown): 'passed' | 'failed' | 'manual-review' {
|
||||
if (!entry || typeof entry !== 'object') return 'failed';
|
||||
if ('manual_review' in entry) return Object.hasOwn(entry, 'manual_review') && isManualReviewEntry(entry) ? 'manual-review' : 'failed';
|
||||
const result = entry as EvalTestEntry;
|
||||
if (result.execution !== undefined && result.execution !== 'executed' && result.execution !== 'reused') return 'failed';
|
||||
return result.passed === true ? 'passed' : 'failed';
|
||||
}
|
||||
|
||||
export interface EvalResult {
|
||||
schema_version: number;
|
||||
version: string;
|
||||
@@ -135,6 +146,7 @@ export interface EvalResult {
|
||||
total_tests: number;
|
||||
executed_tests?: number;
|
||||
reused_tests?: number;
|
||||
manual_accepted_tests?: number;
|
||||
passed: number;
|
||||
failed: number;
|
||||
total_cost_usd: number;
|
||||
@@ -154,10 +166,10 @@ export interface EvalResult {
|
||||
export interface TestDelta {
|
||||
name: string;
|
||||
before: { passed: boolean; cost_usd: number; turns_used?: number; duration_ms?: number;
|
||||
detection_rate?: number; tool_summary?: Record<string, number> };
|
||||
detection_rate?: number; tool_summary?: Record<string, number>; manual_review?: boolean };
|
||||
after: { passed: boolean; cost_usd: number; turns_used?: number; duration_ms?: number;
|
||||
detection_rate?: number; tool_summary?: Record<string, number> };
|
||||
status_change: 'improved' | 'regressed' | 'unchanged';
|
||||
detection_rate?: number; tool_summary?: Record<string, number>; manual_review?: boolean };
|
||||
status_change: 'improved' | 'regressed' | 'unchanged' | 'manual-review';
|
||||
}
|
||||
|
||||
export interface ComparisonResult {
|
||||
@@ -173,6 +185,7 @@ export interface ComparisonResult {
|
||||
improved: number;
|
||||
regressed: number;
|
||||
unchanged: number;
|
||||
manual_reviewed?: number;
|
||||
tool_count_before: number;
|
||||
tool_count_after: number;
|
||||
/** After-tests that had a same-named entry in the before run. 0 = nothing was
|
||||
@@ -388,6 +401,7 @@ export function compareEvalResults(
|
||||
): ComparisonResult {
|
||||
const deltas: TestDelta[] = [];
|
||||
let improved = 0, regressed = 0, unchanged = 0;
|
||||
let manualReviewed = 0;
|
||||
let toolCountBefore = 0, toolCountAfter = 0;
|
||||
let matched = 0;
|
||||
|
||||
@@ -409,33 +423,40 @@ export function compareEvalResults(
|
||||
toolCountAfter += afterToolCount;
|
||||
|
||||
let statusChange: TestDelta['status_change'] = 'unchanged';
|
||||
const beforeManual = beforeTest !== undefined && evalEntryOutcome(beforeTest) === 'manual-review';
|
||||
const afterOutcome = evalEntryOutcome(afterTest);
|
||||
const afterManual = afterOutcome === 'manual-review';
|
||||
if (beforeTest) {
|
||||
matched++;
|
||||
if (!beforeTest.passed && afterTest.passed) { statusChange = 'improved'; improved++; }
|
||||
else if (beforeTest.passed && !afterTest.passed) { statusChange = 'regressed'; regressed++; }
|
||||
if (beforeManual && afterOutcome === 'failed') { statusChange = 'regressed'; regressed++; }
|
||||
else if (beforeManual || afterManual) { statusChange = 'manual-review'; manualReviewed++; }
|
||||
else if (evalEntryOutcome(beforeTest) === 'failed' && evalEntryOutcome(afterTest) === 'passed') { statusChange = 'improved'; improved++; }
|
||||
else if (evalEntryOutcome(beforeTest) === 'passed' && evalEntryOutcome(afterTest) === 'failed') { statusChange = 'regressed'; regressed++; }
|
||||
else { unchanged++; }
|
||||
} else {
|
||||
// New test — treat as unchanged (no prior data)
|
||||
unchanged++;
|
||||
if (afterManual) { statusChange = 'manual-review'; manualReviewed++; }
|
||||
else unchanged++;
|
||||
}
|
||||
|
||||
deltas.push({
|
||||
name: afterTest.name,
|
||||
before: {
|
||||
passed: beforeTest?.passed ?? false,
|
||||
passed: beforeTest !== undefined && evalEntryOutcome(beforeTest) === 'passed',
|
||||
cost_usd: beforeTest?.cost_usd ?? 0,
|
||||
turns_used: beforeTest?.turns_used,
|
||||
duration_ms: beforeTest?.duration_ms,
|
||||
detection_rate: beforeTest?.detection_rate,
|
||||
tool_summary: beforeToolSummary,
|
||||
...(beforeManual ? { manual_review: true } : {}),
|
||||
},
|
||||
after: {
|
||||
passed: afterTest.passed,
|
||||
passed: afterOutcome === 'passed',
|
||||
cost_usd: afterTest.cost_usd,
|
||||
turns_used: afterTest.turns_used,
|
||||
duration_ms: afterTest.duration_ms,
|
||||
detection_rate: afterTest.detection_rate,
|
||||
tool_summary: afterToolSummary,
|
||||
...(afterManual ? { manual_review: true } : {}),
|
||||
},
|
||||
status_change: statusChange,
|
||||
});
|
||||
@@ -452,12 +473,13 @@ export function compareEvalResults(
|
||||
deltas.push({
|
||||
name: `${name} (removed)`,
|
||||
before: {
|
||||
passed: beforeTest.passed,
|
||||
passed: evalEntryOutcome(beforeTest) === 'passed',
|
||||
cost_usd: beforeTest.cost_usd,
|
||||
turns_used: beforeTest.turns_used,
|
||||
duration_ms: beforeTest.duration_ms,
|
||||
detection_rate: beforeTest.detection_rate,
|
||||
tool_summary: beforeToolSummary,
|
||||
...(evalEntryOutcome(beforeTest) === 'manual-review' ? { manual_review: true } : {}),
|
||||
},
|
||||
after: { passed: false, cost_usd: 0, tool_summary: {} },
|
||||
status_change: 'unchanged',
|
||||
@@ -477,6 +499,7 @@ export function compareEvalResults(
|
||||
improved,
|
||||
regressed,
|
||||
unchanged,
|
||||
...(manualReviewed ? { manual_reviewed: manualReviewed } : {}),
|
||||
tool_count_before: toolCountBefore,
|
||||
tool_count_after: toolCountAfter,
|
||||
matched,
|
||||
@@ -495,8 +518,8 @@ export function formatComparison(c: ComparisonResult): string {
|
||||
// Per-test deltas
|
||||
for (const d of c.deltas) {
|
||||
const arrow = d.status_change === 'improved' ? '↑' : d.status_change === 'regressed' ? '↓' : '=';
|
||||
const beforeStatus = d.before.passed ? 'PASS' : 'FAIL';
|
||||
const afterStatus = d.after.passed ? 'PASS' : 'FAIL';
|
||||
const beforeStatus = d.before.manual_review ? 'MANUAL' : d.before.passed ? 'PASS' : 'FAIL';
|
||||
const afterStatus = d.after.manual_review ? 'MANUAL' : d.after.passed ? 'PASS' : 'FAIL';
|
||||
|
||||
// Turns delta
|
||||
let turnsDelta = '';
|
||||
@@ -540,6 +563,7 @@ export function formatComparison(c: ComparisonResult): string {
|
||||
if (c.improved > 0) parts.push(`${c.improved} improved`);
|
||||
if (c.regressed > 0) parts.push(`${c.regressed} regressed`);
|
||||
if (c.unchanged > 0) parts.push(`${c.unchanged} unchanged`);
|
||||
if (c.manual_reviewed) parts.push(`${c.manual_reviewed} unscored manual review`);
|
||||
lines.push(` Status: ${parts.join(', ')}`);
|
||||
|
||||
const costSign = c.total_cost_delta >= 0 ? '+' : '';
|
||||
@@ -607,7 +631,9 @@ export function generateCommentary(c: ComparisonResult): string[] {
|
||||
const regressions = c.deltas.filter(d => d.status_change === 'regressed');
|
||||
if (regressions.length > 0) {
|
||||
for (const d of regressions) {
|
||||
notes.push(`REGRESSION: "${d.name}" was passing, now fails. Investigate immediately.`);
|
||||
notes.push(d.before.manual_review
|
||||
? `REGRESSION: "${d.name}" lost its unscored manual acceptance and now has a blocking failure. Investigate immediately.`
|
||||
: `REGRESSION: "${d.name}" was passing, now fails. Investigate immediately.`);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -617,6 +643,10 @@ export function generateCommentary(c: ComparisonResult): string[] {
|
||||
notes.push(`Fixed: "${d.name}" now passes.`);
|
||||
}
|
||||
|
||||
for (const d of c.deltas.filter(delta => delta.status_change === 'manual-review')) {
|
||||
notes.push(`"${d.name}" includes an unscored manual acceptance; no model-score improvement or regression is inferred.`);
|
||||
}
|
||||
|
||||
// 3. Per-test efficiency changes (only for unchanged-status tests — regressions/improvements are already noted)
|
||||
const stable = c.deltas.filter(d => d.status_change === 'unchanged' && d.after.passed);
|
||||
for (const d of stable) {
|
||||
@@ -893,7 +923,8 @@ export class EvalCollector {
|
||||
const version = getVersion();
|
||||
const totalCost = this.tests.reduce((s, t) => s + t.cost_usd, 0);
|
||||
const totalDuration = this.tests.reduce((s, t) => s + t.duration_ms, 0);
|
||||
const passed = this.tests.filter(t => t.passed).length;
|
||||
const passed = this.tests.filter(t => evalEntryOutcome(t) === 'passed').length;
|
||||
const manual = this.tests.filter(t => evalEntryOutcome(t) === 'manual-review').length;
|
||||
|
||||
const partial: EvalResult = {
|
||||
schema_version: SCHEMA_VERSION,
|
||||
@@ -907,8 +938,9 @@ export class EvalCollector {
|
||||
total_tests: this.tests.length,
|
||||
executed_tests: this.tests.filter(t => t.execution !== 'reused').length,
|
||||
reused_tests: this.tests.filter(t => t.execution === 'reused').length,
|
||||
...(manual ? { manual_accepted_tests: manual } : {}),
|
||||
passed,
|
||||
failed: this.tests.length - passed,
|
||||
failed: this.tests.length - passed - manual,
|
||||
total_cost_usd: Math.round(totalCost * 100) / 100,
|
||||
total_duration_ms: totalDuration,
|
||||
tests: this.tests,
|
||||
@@ -933,7 +965,8 @@ export class EvalCollector {
|
||||
const timestamp = new Date().toISOString();
|
||||
const totalCost = this.tests.reduce((s, t) => s + t.cost_usd, 0);
|
||||
const totalDuration = this.tests.reduce((s, t) => s + t.duration_ms, 0);
|
||||
const passed = this.tests.filter(t => t.passed).length;
|
||||
const passed = this.tests.filter(t => evalEntryOutcome(t) === 'passed').length;
|
||||
const manual = this.tests.filter(t => evalEntryOutcome(t) === 'manual-review').length;
|
||||
|
||||
const flaky = this.flakyRetries();
|
||||
const result: EvalResult = {
|
||||
@@ -948,8 +981,9 @@ export class EvalCollector {
|
||||
total_tests: this.tests.length,
|
||||
executed_tests: this.tests.filter(t => t.execution !== 'reused').length,
|
||||
reused_tests: this.tests.filter(t => t.execution === 'reused').length,
|
||||
...(manual ? { manual_accepted_tests: manual } : {}),
|
||||
passed,
|
||||
failed: this.tests.length - passed,
|
||||
failed: this.tests.length - passed - manual,
|
||||
total_cost_usd: Math.round(totalCost * 100) / 100,
|
||||
total_duration_ms: totalDuration,
|
||||
wall_clock_ms: Date.now() - this.createdAt,
|
||||
@@ -999,7 +1033,9 @@ export class EvalCollector {
|
||||
lines.push('═'.repeat(70));
|
||||
|
||||
for (const t of this.tests) {
|
||||
const status = !t.passed ? ' FAIL ' : t.execution === 'reused' ? ' REUSE' : ' PASS ';
|
||||
const outcome = evalEntryOutcome(t);
|
||||
const status = outcome === 'manual-review' ? 'MANUAL' : outcome === 'failed' ? ' FAIL '
|
||||
: t.execution === 'reused' ? ' REUSE' : ' PASS ';
|
||||
const cost = `$${t.cost_usd.toFixed(2)}`;
|
||||
const dur = t.duration_ms ? `${Math.round(t.duration_ms / 1000)}s` : '';
|
||||
const turns = t.turns_used !== undefined ? `${t.turns_used}t` : '';
|
||||
@@ -1010,6 +1046,8 @@ export class EvalCollector {
|
||||
} else if (t.judge_scores) {
|
||||
const scores = Object.entries(t.judge_scores).map(([k, v]) => `${k[0]}:${v}`).join(' ');
|
||||
detail = scores;
|
||||
} else if (outcome === 'manual-review') {
|
||||
detail = `unscored; approved by ${t.manual_review!.approval.approved_by} (${t.manual_review!.approval.approval_url})`;
|
||||
}
|
||||
|
||||
const name = t.name.length > 35 ? t.name.slice(0, 32) + '...' : t.name.padEnd(35);
|
||||
@@ -1020,6 +1058,7 @@ export class EvalCollector {
|
||||
const totalCost = `$${result.total_cost_usd.toFixed(2)}`;
|
||||
const totalDur = `${Math.round(result.total_duration_ms / 1000)}s`;
|
||||
lines.push(` Total: ${result.passed}/${result.total_tests} passed${' '.repeat(20)}${totalCost.padStart(6)} ${totalDur}`);
|
||||
if (result.manual_accepted_tests) lines.push(` Manual accepted: ${result.manual_accepted_tests} unscored provider refusal(s)`);
|
||||
lines.push(` Evidence: ${result.executed_tests ?? result.total_tests} executed, ${result.reused_tests ?? 0} reused`);
|
||||
if (result.flaky_retries && result.flaky_retries.length > 0) {
|
||||
// Loud, never fatal: a flaky pass must not block anyone, but it must
|
||||
|
||||
@@ -13,7 +13,8 @@ import Anthropic from '@anthropic-ai/sdk';
|
||||
import type { JSONOutputFormat } from '@anthropic-ai/sdk/resources/messages';
|
||||
import { setTimeout as delay } from 'node:timers/promises';
|
||||
|
||||
import { CLAUDE_FRONTIER_EVAL_MODEL, resolveEvalModel } from '../../lib/eval-model';
|
||||
import { CLAUDE_FRONTIER_EVAL_MODEL, DEFAULT_JUDGE_MAX_TOKENS, resolveEvalModel } from '../../lib/eval-model';
|
||||
export { DEFAULT_JUDGE_MAX_TOKENS } from '../../lib/eval-model';
|
||||
|
||||
export interface JudgeScore {
|
||||
clarity: number; // 1-5
|
||||
@@ -22,6 +23,35 @@ export interface JudgeScore {
|
||||
reasoning: string;
|
||||
}
|
||||
|
||||
export interface JudgeRefusalEvidence {
|
||||
stop_reason: 'refusal';
|
||||
response_id: string | null;
|
||||
request_id: string | null;
|
||||
model: string | null;
|
||||
input_tokens: number | null;
|
||||
output_tokens: number | null;
|
||||
text_blocks: number;
|
||||
}
|
||||
|
||||
export class JudgeRefusalError extends Error {
|
||||
readonly refusal: JudgeRefusalEvidence;
|
||||
|
||||
constructor(response: { id?: unknown; _request_id?: unknown; model?: unknown;
|
||||
usage?: { input_tokens?: unknown; output_tokens?: unknown }; content: Array<{ type: string }> }) {
|
||||
super('Judge provider refused the evaluation; no automated score');
|
||||
this.name = 'JudgeRefusalError';
|
||||
this.refusal = {
|
||||
stop_reason: 'refusal',
|
||||
response_id: typeof response.id === 'string' ? response.id : null,
|
||||
request_id: typeof response._request_id === 'string' ? response._request_id : null,
|
||||
model: typeof response.model === 'string' ? response.model : null,
|
||||
input_tokens: typeof response.usage?.input_tokens === 'number' ? response.usage.input_tokens : null,
|
||||
output_tokens: typeof response.usage?.output_tokens === 'number' ? response.usage.output_tokens : null,
|
||||
text_blocks: response.content.filter(block => block.type === 'text').length,
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
export interface OutcomeJudgeResult {
|
||||
detected: string[];
|
||||
missed: string[];
|
||||
@@ -89,7 +119,7 @@ export async function callJudge<T>(
|
||||
// Thinking and answer text share max_tokens. The old 1024-token budget
|
||||
// could be exhausted before a frontier judge emitted any JSON.
|
||||
const resolvedModel = resolveEvalModel('judge', model);
|
||||
const maxTokens = opts?.max_tokens ?? 8192;
|
||||
const maxTokens = opts?.max_tokens ?? DEFAULT_JUDGE_MAX_TOKENS;
|
||||
const client = new Anthropic();
|
||||
|
||||
const makeRequest = () => client.messages.create({
|
||||
@@ -134,6 +164,7 @@ export async function callJudge<T>(
|
||||
.map(block => block.text)
|
||||
.join('\n');
|
||||
try {
|
||||
if (response.stop_reason === 'refusal') throw new JudgeRefusalError(response);
|
||||
if (opts?.jsonSchema !== undefined) {
|
||||
if (response.stop_reason !== 'end_turn') throw new Error(`Structured judge did not complete: stop_reason=${response.stop_reason}`);
|
||||
return JSON.parse(text) as T;
|
||||
|
||||
@@ -0,0 +1,17 @@
|
||||
import { readFileSync } from 'node:fs';
|
||||
import { resolve } from 'node:path';
|
||||
import type { EvalTestEntry } from './eval-store';
|
||||
import { buildCookieWorkflowJudgeInput } from './cookie-workflow-judge-input';
|
||||
import { COOKIE_MANUAL_REVIEW_FILE } from './cookie-workflow-manual-review';
|
||||
|
||||
export function manualReviewFixture(root = resolve(import.meta.dir, '../..')): EvalTestEntry {
|
||||
const approval = JSON.parse(readFileSync(resolve(root, COOKIE_MANUAL_REVIEW_FILE), 'utf8'));
|
||||
return {
|
||||
name: 'setup-browser-cookies/SKILL.md workflow', suite: 'Cookie setup workflow quality', tier: 'llm-judge',
|
||||
passed: false, execution: 'executed', exit_reason: 'provider_refusal', attempt: 1, duration_ms: 1, cost_usd: 0,
|
||||
model: approval.model, prompt: buildCookieWorkflowJudgeInput(root).prompt,
|
||||
error: 'Synthetic provider refusal fixture, not live model evidence',
|
||||
manual_review: { approval, refusal: { stop_reason: 'refusal', response_id: 'msg_synthetic_fixture',
|
||||
request_id: 'req_synthetic_fixture', model: approval.model, input_tokens: 1, output_tokens: 0, text_blocks: 0 } },
|
||||
};
|
||||
}
|
||||
@@ -830,17 +830,31 @@ export type SharedQuestionSelector = (input: Record<string, unknown>) => Record<
|
||||
/** The skip actor may decline work, never approve a mixed fix/preservation choice. */
|
||||
function skippedReviewOption(question: any): any {
|
||||
const options = Array.isArray(question?.options) ? question.options : [];
|
||||
const qualifiedIndexState = /^(?:(?:git\s+)?index|skip-worktree|assume-unchanged)\s+(?:bits?|flags?|attributes?|settings?)$/i;
|
||||
const candidates = options.flatMap((option: any) => {
|
||||
if (typeof option?.label !== 'string' || ['description', 'preview'].some(field =>
|
||||
option[field] !== undefined && typeof option[field] !== 'string')) return [];
|
||||
const label = option.label.replace(/[‘’]/g, "'").replace(/^\s*(?:[A-Z]|\d+)[.)]\s*/i, '')
|
||||
.replace(/\s*\(recommended\)\s*$/i, '').trim();
|
||||
const label = option.label.replace(/[‘’]/g, "'").replace(/`/g, '').replace(/^\s*(?:[A-Z]|\d+)[.)]\s*/i, '')
|
||||
.replace(/\s*\(recommended\)\s*$/i, '').trim()
|
||||
.replace(/^no\s*[,.:!?]\s*(?=(?:skip|decline|keep|leave|do not|don't)\b)/i, '');
|
||||
const referentialRetention = /^(?:keep|leave)\s+(?:it|this|that|them|these)$/i.test(label);
|
||||
const preservation = option.description?.trim().replace(/`/g, '').match(/^(?:keep|leave|retain|preserve)\s+([^,;.!?]+)/i);
|
||||
const preservedObject = preservation?.[1].split(/\b(?:and|but|while)\b/i)[0]
|
||||
.replace(/(?<![-\w])(?:the|this|that|current|existing|local|as[- ]is|unchanged|untouched|set|hidden)\b/gi, '').trim();
|
||||
const describedRetention = !!preservedObject
|
||||
&& !/^\w+ing\b/i.test(preservedObject)
|
||||
&& (/^(?:(?:duplicated|original|prior|tracked|untracked)\s+)*(?:(?:index|skip-worktree|assume-unchanged)\s+)?(?:flags?|code|source|implementations?|copies|copy|files?|routes?|workers?|helpers?|parsers?|changes?|contents?|state|branches|branch|worktrees?)$/i.test(preservedObject)
|
||||
|| qualifiedIndexState.test(preservedObject));
|
||||
const description = (option.description ?? '').replace(/[‘’]/g, "'").trim();
|
||||
const declinesChange = /^(?:do not|don't)\s+(?:apply|change|edit|fix|refactor|extract|modify|touch|clear|remove|update|replace|add|migrate|implement|reuse|import)\b/i;
|
||||
const labelObject = label.match(/^(?:keep|leave)\s+(?:the\s+)?(.+)$/i)?.[1]
|
||||
.replace(/\s+(?:as[- ]is|unchanged|untouched|set)$/i, '');
|
||||
const preservationRank = referentialRetention ? describedRetention || declinesChange.test(description)
|
||||
: /^(?:keep|leave)\b.*\b(?:current|existing|unchanged|untouched|as[- ]is|alone|set|copies|copy|implementation|code|source)\b/i.test(label)
|
||||
|| !!labelObject && qualifiedIndexState.test(labelObject);
|
||||
const rank = /^(?:skip|decline)(?=$|\s|[,.!])/i.test(label) ? 3
|
||||
: declinesChange.test(label) ? 2
|
||||
: /^(?:keep|leave)\b.*\b(?:current|existing|unchanged|untouched|as[- ]is|alone|set|copies|copy|implementation|code|source)\b/i.test(label)
|
||||
|| (/^(?:keep|leave)\b/i.test(label) && declinesChange.test(description)) ? 1 : 0;
|
||||
: preservationRank ? 1 : 0;
|
||||
if (!rank) return [];
|
||||
// A leading decline names rejected work. Classify later commitments rather
|
||||
// than action words inside recorded metadata or hypothetical consequences.
|
||||
@@ -848,21 +862,34 @@ function skippedReviewOption(question: any): any {
|
||||
option.description ?? '', option.preview ?? ''].join('\n').replace(/[‘’]/g, "'");
|
||||
const actions = new Set(['approve', 'fix', 'apply', 'refactor', 'extract', 'replace', 'rewrite', 'edit', 'modify',
|
||||
'change', 'clear', 'remove', 'delete', 'add', 'update', 'implement', 'migrate', 'touch', 're-export',
|
||||
'import', 'reuse', 'share', 'wire', 'convert']);
|
||||
'import', 'reuse', 'share', 'wire', 'convert', 'set', 'unset', 'toggle', 'flip', 'reset', 'enable', 'disable']);
|
||||
const isAction = (word = '') => [word, word.replace(/s$/, ''), word.replace(/(?:es|ed|ing)$/, ''),
|
||||
word.replace(/(?:ed|ing)$/, 'e'), word.replace(/(?:ies|ied)$/, 'y')].some(form => actions.has(form));
|
||||
word.replace(/(?:ed|ing)$/, 'e'), word.replace(/(?:ies|ied)$/, 'y'),
|
||||
word.replace(/([a-z])\1(?:ed|ing)$/, '$1')].some(form => actions.has(form));
|
||||
const changes = commitment.toLowerCase().split(/[,;\n]|[.!?](?:\s|$)|\b(?:and|but|then|while)\b/).some(part => {
|
||||
const clause = part.replace(/^[^a-z]+/, '')
|
||||
.replace(/^(?:(?:this|that|the|selected|chosen)\s+(?:option|choice|selection)|i|we|you|it|(?:the\s+)?(?:source|code|route|worker|helper|parser|index(?:\s+flag)?))\s+/, '')
|
||||
.replace(/^(?:will|would|should|must|can|may|does|do)\s+/, '')
|
||||
.replace(/^(?:(?:please|also|still|just|now|be)\s+)+/, '');
|
||||
if (/^(?:not|does not|don't|doesn't|won't|without|no)\b/.test(clause)) return false;
|
||||
// The no-change choice may persist/reuse its review decision. That is not
|
||||
// permission to modify source or clear an index flag.
|
||||
if (/^(?:updates?|updated|updating|reuses?|reused|reusing)\s+(?:the\s+)?(?:(?:prior|recorded|existing)\s+)?(?:review\s+(?:log|record)|decision|advisory|snapshot|ledger)\b/.test(clause)) return false;
|
||||
const first = clause.match(/^[a-z]+(?:-[a-z]+)*/)?.[0];
|
||||
const future = clause.match(/\bwill\s+(?:be\s+)?([a-z]+(?:-[a-z]+)*)/)?.[1];
|
||||
return isAction(first) || isAction(future);
|
||||
const futureMatch = clause.match(/\b(?:will|would|should|must|can|may)\s+(?:(?:still|also|now|just|[a-z]+ly)\s+)*(?:be\s+)?(?:(?:still|also|now|just|[a-z]+ly)\s+)*([a-z]+(?:-[a-z]+)*)/);
|
||||
const future = futureMatch?.[1];
|
||||
const method = /^(?:keep|leave|retain|preserve)\b/.test(clause)
|
||||
&& [...clause.matchAll(/\b(?:by|through|via)\s+([a-z]+(?:-[a-z]+)*)/g)].some(match => isAction(match[1]));
|
||||
const state = /^(?:change|modification|file|flag|state|content)\s+(?:stays?|remains?)\b/.test(clause);
|
||||
const recordedDecision = /^(?:updates?|updated|updating|reuses?|reused|reusing)\s+(?:the\s+)?(?:(?:prior|recorded|existing)\s+)?(?:review\s+(?:log|record)|decision|advisory|snapshot|ledger)\b/.test(clause);
|
||||
const nominalReuse = /^(?:the\s+)?reuse\s+of\b/.test(clause);
|
||||
const describedReuse = /\b(?:is|are|was|were|remains?|stays?|requires?|needs?|will|would|should|must|can|may)\s+(?:(?:still|also|now|just|not|never|[a-z]+ly)\s+)*[a-z]+(?:-[a-z]+)*/.test(clause);
|
||||
const futureSubject = clause.slice(0, futureMatch?.index ?? 0).trim();
|
||||
const passiveDecision = /\b(?:review\s+(?:log|record)|decision|advisory|snapshot|ledger)$/.test(futureSubject)
|
||||
|| /\b(?:review\s+(?:log|record)|decision|advisory|snapshot|ledger)\b(?:(?!\b(?:source|code|route|worker|helper|parser|file|flag)\b).)*\bit$/.test(futureSubject);
|
||||
const futureDecision = /\b(?:can|will|would|should|must|may)\s+(?:(?:still|also|now|just|[a-z]+ly)\s+)*reuse\s+(?:(?:this|the|prior|recorded|existing)\s+)*(?:review\s+(?:log|record)|decision|advisory|snapshot|ledger)\b/.test(clause);
|
||||
const purpose = [...clause.matchAll(/\b(?:to|by|through|via)\s+(?:[a-z]+ly\s+)*([a-z]+(?:-[a-z]+)*)/g)]
|
||||
.some(match => isAction(match[1]));
|
||||
return (isAction(future) && !(future === 'reused' && passiveDecision) && !(future === 'reuse' && futureDecision)) || method || purpose
|
||||
|| (nominalReuse && !describedReuse)
|
||||
|| (!state && !recordedDecision && !nominalReuse && isAction(first));
|
||||
});
|
||||
return changes ? [] : [{ option, rank }];
|
||||
});
|
||||
|
||||
@@ -1772,6 +1772,7 @@ export const E2E_TIERS: Record<string, 'gate' | 'periodic'> = {
|
||||
* LLM-judge test touchfiles — keyed by test description string.
|
||||
*/
|
||||
export const LLM_JUDGE_TOUCHFILES: Record<string, string[]> = {
|
||||
'setup-browser-cookies/SKILL.md workflow': ['setup-browser-cookies/SKILL.md.tmpl', 'setup-browser-cookies/SKILL.md', 'BROWSER.md', 'test/helpers/cookie-workflow-judge-input.ts', 'test/cookie-workflow-judge-input.test.ts', 'test/helpers/cookie-workflow-manual-review.ts', 'test/cookie-workflow-manual-review.test.ts', 'test/helpers/manual-judge-review-fixture.ts', '.github/cookie-workflow-manual-review.json', 'test/helpers/workflow-judge-input.ts', 'test/skill-llm-eval.test.ts'],
|
||||
'command reference table': ['browse/sections/**', 'SKILL.md', 'SKILL.md.tmpl', 'browse/src/commands.ts', 'test/skill-llm-eval.test.ts'],
|
||||
'snapshot flags reference': ['browse/sections/**', 'SKILL.md', 'SKILL.md.tmpl', 'browse/src/snapshot.ts', 'test/skill-llm-eval.test.ts'],
|
||||
'browse/SKILL.md reference': ['browse/sections/**', 'browse/SKILL.md', 'browse/SKILL.md.tmpl', 'browse/src/**', 'test/skill-llm-eval.test.ts'],
|
||||
|
||||
@@ -3,7 +3,7 @@ import * as fs from 'node:fs';
|
||||
import * as path from 'node:path';
|
||||
import { isBuiltin } from 'node:module';
|
||||
import { spawnSync } from 'node:child_process';
|
||||
import { resolveEvalModel } from '../../lib/eval-model';
|
||||
import { DEFAULT_JUDGE_MAX_TOKENS, resolveEvalModel } from '../../lib/eval-model';
|
||||
import { JUDGE_MS } from './eval-budgets';
|
||||
import type { JudgeScore } from './llm-judge';
|
||||
import { readWorkflowJudgeInput, buildWorkflowJudgePrompt } from './workflow-judge-input';
|
||||
@@ -98,7 +98,7 @@ export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): {
|
||||
coverage: { dependencies: 'complete', prompts: 'complete', environment: 'complete' }, unknownDependencies: [],
|
||||
files: workflowJudgeDependencies(opts.root, input.files.map(file => file.path)),
|
||||
prompts: { [opts.testName]: prompt },
|
||||
parameters: { rootPackage, thresholds: opts.thresholds, max_tokens: 8192, temperature: null, budget_ms: JUDGE_MS,
|
||||
parameters: { rootPackage, thresholds: opts.thresholds, max_tokens: DEFAULT_JUDGE_MAX_TOKENS, temperature: null, budget_ms: JUDGE_MS,
|
||||
request: 'messages.create/user', retries: 1 },
|
||||
runtime: { image: env.EVALS_CACHE_RUNTIME_ID!, bun: Bun.version, node: process.versions.node,
|
||||
platform: process.platform, arch: process.arch, judge: resolveEvalModel('judge', undefined, env),
|
||||
|
||||
Reference in New Issue
Block a user