v1.90.0.0 feat: make browser cookie imports explicit and safe (#2964)

* fix(browse): prepare reliable cookie import wave for validation

* ci: sequence quality and behavior for validation branch

* fix(browse): isolate Windows qualification and preserve native diagnostics

* test(browse): cover cookie workflow quality and isolate Windows user paths

* test(browse): trace native member startup and initialize fresh folders

* fix(browse): keep Windows member stdin alive through EOF

* fix(browse): latch native timeouts and compare contained Edge startup

* test(browse): verify native version metadata and actual Windows argv

* test(browse): qualify Dia import on isolated macOS CI

* fix(browse): require picker origin for session mutations

* fix(browse): bound credential reads through stream completion

* test(browse): inspect owned Windows process arguments natively

* test(evals): preserve passing coverage during cookie repair reruns

* test(browse): isolate Dia qualification in a fresh macOS account

* test(browse): pass bounded integer timeouts to native Mac probes

* test(browse): distinguish Windows profile initialization from containment

* test(browse): await descendant pipe readiness before parent exit

* test(browse): initialize and restore isolated macOS Keychain state

* test(browse): initialize Windows fixture folders before qualification

* test(ci): pin the same Node runtime across Windows checks

* test(browse): distinguish native macOS browser preflight stages

* test(browse): isolate Windows descendant console lifetime

* test(browse): preserve native receipts and identify fixture lock holders

* test(browse): prepare dependency resolution before native Mac worker startup

* test(ci): include lock and close checks in native diagnostics

* test(browse): preserve native owner probe stages and subprocess deadlines

* fix(browse): classify Chromium profile-in-use exit precisely

* test(browse): retain Mac qualification evidence through cleanup failures

* test(browse): bound Mac fixture paths and retire its owned user domain

* test(browse): accept vanished fixture entries without weakening cleanup

* test(browse): identify probe-created macOS user domains safely

* test(browse): observe Mac user domains without targeting them first

* test(browse): use passive fresh-user ownership throughout Mac qualification

* test(browse): distinguish profile and registered-home Keychain lookups

* test(browse): qualify Dia under one registered account home

* test(browse): identify Dia startup and owned process-group failures

* test(browse): classify bounded Dia startup diagnostics without leaking output

* fix(test): preserve native Mac sandboxing and reap owned browser children

* fix(browse): preserve Chromium sandboxing for native profile imports

* test(browse): inspect signed Mach-O architecture without launching Xcode tools

* test(browse): sample pending Dia startup and reap on all cleanup paths

* test(browse): compare protected Dia launches in fresh Bun and Node accounts

* test(browse): inspect isolated Mac GUI readiness without browser access

* v1.90.0.0 fix: bind cookie picker actions to their document

* test: validate cookie guards and fit nested launch fixtures

* ci: configure the bundled Chromium sandbox helper

* fix(browse): classify Playwright authentication timeouts

* test: retain bounded Windows lifecycle diagnostics

* test(cso): reuse bounded NTFS precision candidates

* test(review): handle explicit preservation choices safely

* test(browse): remove owned fixture directories with explicit primitives

* test(review): distinguish descriptive reuse from edit commitments

* test: admit only the approved unscored cookie workflow refusal

* test: keep the Office Hours judge mock export-complete

* fix: keep dependency-free CI planners independent of the model SDK

* test: observe the exact holder after a native fixture unlink failure

* fix: start seeded PTY observations at owned readiness

* test: acquire identity-bound Windows deletion admission before profile resets

* test: preserve qualified Git index bits without authorizing mutations
This commit is contained in:
Garry Tan
2026-09-25 12:06:45 -04:00
committed by GitHub
parent 730a1017d1
commit a84b0b5b6d
111 changed files with 14996 additions and 1057 deletions
+61
View File
@@ -29,11 +29,22 @@ import * as os from 'os';
import * as path from 'path';
import { runBin } from './helpers/run-bin';
import { selectTests, E2E_TOUCHFILES, LLM_JUDGE_TOUCHFILES, GLOBAL_TOUCHFILES } from './helpers/touchfiles';
import { manualReviewFixture } from './helpers/manual-judge-review-fixture';
import { renderDashboard } from '../scripts/eval-watch';
const ROOT = path.resolve(import.meta.dir, '..');
const SCRIPT = (name: string) => path.join(ROOT, 'scripts', name);
const SLUG = 'eval-cli-fixture';
test('eval:watch distinguishes unscored manual acceptance from malformed claims', () => {
const manual = manualReviewFixture();
const output = renderDashboard(null, { tests: [manual, { ...manual, passed: true }], total_cost_usd: 0 });
expect(output).toContain('MANUAL/unscored');
expect(output).toContain(manual.manual_review!.approval.approval_url);
expect(output).toContain('Manual accepted: 1');
expect(output).toContain('✗');
});
let tmpHome: string;
let evalDir: string;
@@ -256,6 +267,22 @@ describe('eval:list CLI (scripts/eval-list.ts)', () => {
// ── eval-compare ─────────────────────────────────────────────────────────────
describe('eval:compare CLI (scripts/eval-compare.ts)', () => {
test('shows a manual transition without calling it a scored regression', () => {
const manual = manualReviewFixture();
const prior = writeRun(evalDir, { tier: 'llm-judge', timestamp: '2026-01-01T01:00:00Z',
tests: [{ name: manual.name, passed: true }] });
const after = writeRun(evalDir, { tier: 'llm-judge', timestamp: '2026-01-02T01:00:00Z',
tests: [{ name: manual.name, passed: false }] });
const body = JSON.parse(fs.readFileSync(after, 'utf8'));
body.tests = [manual]; body.passed = 0; body.failed = 0; body.manual_accepted_tests = 1;
fs.writeFileSync(after, JSON.stringify(body));
const result = runEvalCli('eval-compare.ts', prior, after);
expect(result.status).toBe(0);
expect(result.stdout).toContain('PASS → MANUAL');
expect(result.stdout).toContain('unscored manual review');
expect(result.stdout).not.toContain('REGRESSION:');
});
test('empty eval dir prints the getting-started hint and exits 0', () => {
const result = runEvalCli('eval-compare.ts');
expect(result.status).toBe(0);
@@ -337,6 +364,40 @@ describe('eval:compare CLI (scripts/eval-compare.ts)', () => {
// ── eval-summary ─────────────────────────────────────────────────────────────
describe('eval:summary CLI (scripts/eval-summary.ts)', () => {
test('reports manual provenance without inventing a scored flake', () => {
const manual = manualReviewFixture();
writeRun(evalDir, { tier: 'llm-judge', timestamp: '2026-01-01T01:00:00Z',
tests: [{ name: manual.name, passed: true }] });
const accepted = writeRun(evalDir, { tier: 'llm-judge', timestamp: '2026-01-02T01:00:00Z',
tests: [{ name: manual.name, passed: false }] });
const body = JSON.parse(fs.readFileSync(accepted, 'utf8'));
body.tests = [manual]; body.passed = 0; body.failed = 0; body.manual_accepted_tests = 1;
fs.writeFileSync(accepted, JSON.stringify(body));
const result = runEvalCli('eval-summary.ts');
expect(result.status).toBe(0);
expect(result.stdout).toContain('Manual accepted: 1 unscored provider refusal');
expect(result.stdout).toContain(manual.manual_review!.approval.approval_url);
expect(result.stdout).not.toContain('Flaky tests');
});
test('retried manual claims remain failed in summary history', () => {
const manual = manualReviewFixture();
writeRun(evalDir, { tier: 'llm-judge', timestamp: '2026-01-01T01:00:00Z',
tests: [{ name: manual.name, passed: true }] });
const attempted = writeRun(evalDir, { tier: 'llm-judge', timestamp: '2026-01-02T01:00:00Z',
tests: [{ name: manual.name, passed: false }] });
const body = JSON.parse(fs.readFileSync(attempted, 'utf8'));
const { manual_review: _receipt, ...ordinary } = manual;
body.tests = [{ ...ordinary, attempt: 1 }, { ...manual, attempt: 2 }];
body.total_tests = 2; body.passed = 0; body.failed = 2;
fs.writeFileSync(attempted, JSON.stringify(body));
const result = runEvalCli('eval-summary.ts');
expect(result.status).toBe(0);
expect(result.stdout).not.toContain('Manual accepted:');
expect(result.stdout).toContain('Flaky tests (1):');
expect(result.stdout).toContain(`llm-judge:${manual.name}`);
});
test('empty eval dir prints the getting-started hint and exits 0', () => {
const result = runEvalCli('eval-summary.ts');
expect(result.status).toBe(0);