v1.90.0.0 feat: make browser cookie imports explicit and safe (#2964)

* fix(browse): prepare reliable cookie import wave for validation

* ci: sequence quality and behavior for validation branch

* fix(browse): isolate Windows qualification and preserve native diagnostics

* test(browse): cover cookie workflow quality and isolate Windows user paths

* test(browse): trace native member startup and initialize fresh folders

* fix(browse): keep Windows member stdin alive through EOF

* fix(browse): latch native timeouts and compare contained Edge startup

* test(browse): verify native version metadata and actual Windows argv

* test(browse): qualify Dia import on isolated macOS CI

* fix(browse): require picker origin for session mutations

* fix(browse): bound credential reads through stream completion

* test(browse): inspect owned Windows process arguments natively

* test(evals): preserve passing coverage during cookie repair reruns

* test(browse): isolate Dia qualification in a fresh macOS account

* test(browse): pass bounded integer timeouts to native Mac probes

* test(browse): distinguish Windows profile initialization from containment

* test(browse): await descendant pipe readiness before parent exit

* test(browse): initialize and restore isolated macOS Keychain state

* test(browse): initialize Windows fixture folders before qualification

* test(ci): pin the same Node runtime across Windows checks

* test(browse): distinguish native macOS browser preflight stages

* test(browse): isolate Windows descendant console lifetime

* test(browse): preserve native receipts and identify fixture lock holders

* test(browse): prepare dependency resolution before native Mac worker startup

* test(ci): include lock and close checks in native diagnostics

* test(browse): preserve native owner probe stages and subprocess deadlines

* fix(browse): classify Chromium profile-in-use exit precisely

* test(browse): retain Mac qualification evidence through cleanup failures

* test(browse): bound Mac fixture paths and retire its owned user domain

* test(browse): accept vanished fixture entries without weakening cleanup

* test(browse): identify probe-created macOS user domains safely

* test(browse): observe Mac user domains without targeting them first

* test(browse): use passive fresh-user ownership throughout Mac qualification

* test(browse): distinguish profile and registered-home Keychain lookups

* test(browse): qualify Dia under one registered account home

* test(browse): identify Dia startup and owned process-group failures

* test(browse): classify bounded Dia startup diagnostics without leaking output

* fix(test): preserve native Mac sandboxing and reap owned browser children

* fix(browse): preserve Chromium sandboxing for native profile imports

* test(browse): inspect signed Mach-O architecture without launching Xcode tools

* test(browse): sample pending Dia startup and reap on all cleanup paths

* test(browse): compare protected Dia launches in fresh Bun and Node accounts

* test(browse): inspect isolated Mac GUI readiness without browser access

* v1.90.0.0 fix: bind cookie picker actions to their document

* test: validate cookie guards and fit nested launch fixtures

* ci: configure the bundled Chromium sandbox helper

* fix(browse): classify Playwright authentication timeouts

* test: retain bounded Windows lifecycle diagnostics

* test(cso): reuse bounded NTFS precision candidates

* test(review): handle explicit preservation choices safely

* test(browse): remove owned fixture directories with explicit primitives

* test(review): distinguish descriptive reuse from edit commitments

* test: admit only the approved unscored cookie workflow refusal

* test: keep the Office Hours judge mock export-complete

* fix: keep dependency-free CI planners independent of the model SDK

* test: observe the exact holder after a native fixture unlink failure

* fix: start seeded PTY observations at owned readiness

* test: acquire identity-bound Windows deletion admission before profile resets

* test: preserve qualified Git index bits without authorizing mutations
This commit is contained in:
Garry Tan
2026-09-25 12:06:45 -04:00
committed by GitHub
parent 730a1017d1
commit a84b0b5b6d
111 changed files with 14996 additions and 1057 deletions
+71
View File
@@ -10,11 +10,13 @@ import {
isPartialEval,
listEvalJsonFiles,
compareEvalResults,
evalEntryOutcome,
formatComparison,
generateCommentary,
judgePassed,
} from './eval-store';
import type { EvalResult, EvalTestEntry, ComparisonResult } from './eval-store';
import { manualReviewFixture } from './manual-judge-review-fixture';
let tmpDir: string;
@@ -77,6 +79,36 @@ async function captureStderr(fn: () => Promise<void>): Promise<string> {
// --- EvalCollector tests ---
describe('EvalCollector', () => {
test('manual provider refusal stays unscored and executed; malformed claims stay failed', async () => {
const manual = manualReviewFixture();
const invalidPass = { ...manual, passed: true };
const invalidScore = { ...manual, judge_scores: { clarity: 5 } };
expect(evalEntryOutcome(manual)).toBe('manual-review');
expect(evalEntryOutcome(invalidPass)).toBe('failed');
expect(evalEntryOutcome(invalidScore)).toBe('failed');
expect(evalEntryOutcome({ ...manual, manual_review: { ...manual.manual_review, approval: { ...manual.manual_review!.approval,
prompt_sha256: '0'.repeat(64) } } })).toBe('failed');
const collector = new EvalCollector('llm-judge', tmpDir);
collector.addTest(makeEntry({ name: 'ordinary', tier: 'llm-judge' }));
collector.addTest(manual);
collector.addTest(invalidPass);
collector.addTest(invalidScore);
const partial: EvalResult = JSON.parse(fs.readFileSync(path.join(tmpDir, '_partial-e2e.json'), 'utf8'));
expect(partial).toMatchObject({ passed: 1, failed: 2, manual_accepted_tests: 1,
executed_tests: 4, reused_tests: 0 });
const output = await captureStderr(async () => { await collector.finalize(); });
const result: EvalResult = JSON.parse(fs.readFileSync(fs.readdirSync(tmpDir).map(name => path.join(tmpDir, name))
.find(name => name.endsWith('.json') && !path.basename(name).startsWith('_'))!, 'utf8'));
expect(result).toMatchObject({ passed: 1, failed: 2, manual_accepted_tests: 1, executed_tests: 4 });
expect(result.tests[1]).toMatchObject({ passed: false, execution: 'executed', exit_reason: 'provider_refusal' });
expect(result.tests[1].judge_scores).toBeUndefined();
expect(output).toContain('MANUAL');
expect(output).toContain('1 unscored provider refusal');
expect(output).toContain(manual.manual_review!.approval.approval_url);
expect(result.tests[2].passed).toBe(true);
expect(output).toContain(' FAIL ');
});
test('reused passing evidence preserves origin and remains separate from newly executed attempts', async () => {
const collector = new EvalCollector('llm-judge', tmpDir);
const reused_from = { input_key: 'a'.repeat(64), run_id: '1234/1', revision: 'b'.repeat(40),
@@ -590,6 +622,45 @@ describe('findLatestFinalizedRun', () => {
// --- compareEvalResults tests ---
describe('compareEvalResults', () => {
test('manual acceptance is neither a score regression nor a recovery', () => {
const manual = manualReviewFixture();
const prior = makeResult({ tests: [makeEntry({ name: manual.name, passed: true })] });
const accepted = makeResult({ tests: [manual], passed: 0, failed: 0, manual_accepted_tests: 1 });
const toManual = compareEvalResults(prior, accepted, 'prior.json', 'accepted.json');
expect(toManual).toMatchObject({ improved: 0, regressed: 0, manual_reviewed: 1 });
expect(toManual.deltas[0]).toMatchObject({ status_change: 'manual-review', before: { passed: true },
after: { passed: false, manual_review: true } });
expect(formatComparison(toManual)).toContain('PASS → MANUAL');
expect(formatComparison(toManual)).not.toContain('REGRESSION:');
const fromManual = compareEvalResults(accepted, prior, 'accepted.json', 'prior.json');
expect(fromManual).toMatchObject({ improved: 0, regressed: 0, manual_reviewed: 1 });
expect(formatComparison(fromManual)).toContain('MANUAL → PASS');
const invalid = makeResult({ tests: [{ ...manual, passed: true }] });
expect(compareEvalResults(prior, invalid, 'prior.json', 'invalid.json').regressed).toBe(1);
});
test('manual acceptance followed by a real or malformed failure is a blocking regression', () => {
const manual = manualReviewFixture();
const accepted = makeResult({ tests: [manual], passed: 0, failed: 0, manual_accepted_tests: 1 });
const { manual_review: _receipt, ...ordinary } = manual;
for (const after of [
{ ...ordinary, exit_reason: 'timeout' },
{ ...manual, passed: true },
{ ...manual, judge_scores: { clarity: 5 } },
]) {
const failed = makeResult({ tests: [after] });
const comparison = compareEvalResults(accepted, failed, 'accepted.json', 'failed.json');
expect(comparison).toMatchObject({ improved: 0, regressed: 1 });
expect(comparison.manual_reviewed).toBeUndefined();
expect(comparison.deltas[0]).toMatchObject({ status_change: 'regressed', before: { manual_review: true },
after: { passed: false } });
const output = formatComparison(comparison);
expect(output).toContain('MANUAL → FAIL');
expect(output).toContain('REGRESSION:');
expect(output).toContain('blocking failure');
}
});
test('detects improved/regressed/unchanged per test', () => {
const before = makeResult({
tests: [