v1.90.0.0 feat: make browser cookie imports explicit and safe (#2964)

* fix(browse): prepare reliable cookie import wave for validation

* ci: sequence quality and behavior for validation branch

* fix(browse): isolate Windows qualification and preserve native diagnostics

* test(browse): cover cookie workflow quality and isolate Windows user paths

* test(browse): trace native member startup and initialize fresh folders

* fix(browse): keep Windows member stdin alive through EOF

* fix(browse): latch native timeouts and compare contained Edge startup

* test(browse): verify native version metadata and actual Windows argv

* test(browse): qualify Dia import on isolated macOS CI

* fix(browse): require picker origin for session mutations

* fix(browse): bound credential reads through stream completion

* test(browse): inspect owned Windows process arguments natively

* test(evals): preserve passing coverage during cookie repair reruns

* test(browse): isolate Dia qualification in a fresh macOS account

* test(browse): pass bounded integer timeouts to native Mac probes

* test(browse): distinguish Windows profile initialization from containment

* test(browse): await descendant pipe readiness before parent exit

* test(browse): initialize and restore isolated macOS Keychain state

* test(browse): initialize Windows fixture folders before qualification

* test(ci): pin the same Node runtime across Windows checks

* test(browse): distinguish native macOS browser preflight stages

* test(browse): isolate Windows descendant console lifetime

* test(browse): preserve native receipts and identify fixture lock holders

* test(browse): prepare dependency resolution before native Mac worker startup

* test(ci): include lock and close checks in native diagnostics

* test(browse): preserve native owner probe stages and subprocess deadlines

* fix(browse): classify Chromium profile-in-use exit precisely

* test(browse): retain Mac qualification evidence through cleanup failures

* test(browse): bound Mac fixture paths and retire its owned user domain

* test(browse): accept vanished fixture entries without weakening cleanup

* test(browse): identify probe-created macOS user domains safely

* test(browse): observe Mac user domains without targeting them first

* test(browse): use passive fresh-user ownership throughout Mac qualification

* test(browse): distinguish profile and registered-home Keychain lookups

* test(browse): qualify Dia under one registered account home

* test(browse): identify Dia startup and owned process-group failures

* test(browse): classify bounded Dia startup diagnostics without leaking output

* fix(test): preserve native Mac sandboxing and reap owned browser children

* fix(browse): preserve Chromium sandboxing for native profile imports

* test(browse): inspect signed Mach-O architecture without launching Xcode tools

* test(browse): sample pending Dia startup and reap on all cleanup paths

* test(browse): compare protected Dia launches in fresh Bun and Node accounts

* test(browse): inspect isolated Mac GUI readiness without browser access

* v1.90.0.0 fix: bind cookie picker actions to their document

* test: validate cookie guards and fit nested launch fixtures

* ci: configure the bundled Chromium sandbox helper

* fix(browse): classify Playwright authentication timeouts

* test: retain bounded Windows lifecycle diagnostics

* test(cso): reuse bounded NTFS precision candidates

* test(review): handle explicit preservation choices safely

* test(browse): remove owned fixture directories with explicit primitives

* test(review): distinguish descriptive reuse from edit commitments

* test: admit only the approved unscored cookie workflow refusal

* test: keep the Office Hours judge mock export-complete

* fix: keep dependency-free CI planners independent of the model SDK

* test: observe the exact holder after a native fixture unlink failure

* fix: start seeded PTY observations at owned readiness

* test: acquire identity-bound Windows deletion admission before profile resets

* test: preserve qualified Git index bits without authorizing mutations
This commit is contained in:
Garry Tan
2026-09-25 12:06:45 -04:00
committed by GitHub
parent 730a1017d1
commit a84b0b5b6d
111 changed files with 14996 additions and 1057 deletions
+65 -11
View File
@@ -66,7 +66,8 @@ import {
import { PAID_TEST_GLOBS, isPaidTestFile } from '../test/helpers/paid-test-set';
import { PERIODIC_CI_EXCLUDE } from '../test/helpers/periodic-exclude-data';
import { AUTOPLAN_CHAIN_BUDGET, FILE_RETRY_BUDGETS, STRICT_RETRY_CASE_BUDGETS } from '../test/helpers/eval-budgets';
import { getProjectEvalDir, getClaudeCliVersion, isFinalizedEvalResultFile } from '../test/helpers/eval-store';
import { getProjectEvalDir, getClaudeCliVersion, isFinalizedEvalResultFile, evalEntryOutcome } from '../test/helpers/eval-store';
import { manualReviewProblem } from '../test/helpers/cookie-workflow-manual-review';
import { preflightAnthropicApi } from '../test/helpers/anthropic-preflight';
import { OVERLAY_MIN_FILE_WALL_MS } from '../test/helpers/overlay-case-policy';
import { PR_PROFILE_CASE_IDS, PR_PROFILE_FILES, packageChangeOnlyVersion, selectPrProfile, type PrProfileSelection } from './test-pr-profile';
@@ -1314,19 +1315,20 @@ export function formatProfileCoverage(manifest: PaidRunManifest): string[] {
/** Final outcomes use each case's last attempt; the attempt total stays visible. */
export function collectorOutcomeCounts(results: Array<{ tests?: Array<{
name: string; suite?: string; passed: boolean; execution?: string;
}> }>): { executed: number; reused: number; passed: number; failed: number; attempts: number } {
const counts = { executed: 0, reused: 0, passed: 0, failed: 0, attempts: 0 };
name: string; suite?: string; passed: boolean; execution?: string; manual_review?: unknown;
}> }>): { executed: number; reused: number; passed: number; failed: number; manual_accepted: number; attempts: number } {
const counts = { executed: 0, reused: 0, passed: 0, failed: 0, manual_accepted: 0, attempts: 0 };
for (const result of results) {
const cases = new Map<string, NonNullable<typeof result.tests>[number]>();
for (const entry of result.tests ?? []) {
if (typeof entry.name !== 'string' || typeof entry.passed !== 'boolean') continue;
if (!entry || typeof entry !== 'object' || typeof entry.name !== 'string' || typeof entry.passed !== 'boolean') continue;
counts.attempts++;
cases.set(`${entry.suite ?? ''}\0${entry.name}`, entry);
}
for (const entry of cases.values()) {
counts[entry.execution === 'reused' ? 'reused' : 'executed']++;
counts[entry.passed ? 'passed' : 'failed']++;
const outcome = evalEntryOutcome(entry);
counts[outcome === 'manual-review' ? 'manual_accepted' : outcome]++;
}
}
return counts;
@@ -1479,6 +1481,8 @@ async function main(): Promise<number> {
// ── Report mode: reconcile slice artifacts against the manifest. Fail-closed:
// a slice whose artifact never landed is a FAILURE, not an absence.
if (options.reportDir) {
const summaryPath = path.join(options.reportDir, 'collector-outcomes.json');
fs.rmSync(summaryPath, { force: true });
const manifest = parseRunManifest(fs.readFileSync(path.join(options.reportDir, 'manifest.json'), 'utf-8'));
const results: SliceResult[] = fs.readdirSync(options.reportDir)
.filter((name) => /^slice-\d+\.json$/.test(name))
@@ -1498,16 +1502,54 @@ async function main(): Promise<number> {
// Source: the finalized eval-store JSONs inside the slice artifacts.
const flaky: Array<{ name: string; attempts: number; file: string }> = [];
const collectors: Parameters<typeof collectorOutcomeCounts>[0] = [];
const files: Array<{ file: string; tier: string; shard: string | number; cost: number;
flaky: number; total: number; executed: number; reused: number; passed: number;
failed: number; manual_accepted: number; attempts: number }> = [];
const manualProblems: string[] = [];
const manualClaims = new Map<string, string>();
for (const name of fs.readdirSync(options.reportDir, { recursive: true }) as string[]) {
if (!isFinalizedEvalResultFile(name)) continue;
try {
const parsed = JSON.parse(fs.readFileSync(path.join(options.reportDir, name), 'utf-8'));
if (Array.isArray(parsed.tests)) collectors.push(parsed);
if (!Array.isArray(parsed.tests)) {
if (Object.hasOwn(parsed, 'tests') || parsed.total_tests !== undefined || parsed.manual_review !== undefined) {
manualProblems.push(`${name}: malformed collector tests[]`);
}
continue;
}
const seen = new Map<string, number>();
for (const [index, entry] of parsed.tests.entries()) {
if (!entry || typeof entry !== 'object' || typeof entry.name !== 'string' || !entry.name
|| typeof entry.passed !== 'boolean') {
manualProblems.push(`${name}: attempt ${index + 1}: malformed collector entry (name/passed required)`);
continue;
}
const key = `${entry.suite ?? ''}\0${entry.name}`;
const occurrence = (seen.get(key) ?? 0) + 1;
seen.set(key, occurrence);
if (Object.hasOwn(entry, 'manual_review') && occurrence !== 1) {
manualProblems.push(`${name}: attempt ${index + 1}: manual review is only valid on the first case attempt`);
}
if (Object.hasOwn(entry, 'manual_review')) {
const previous = manualClaims.get(key);
if (previous && previous !== name) manualProblems.push(`${name}: duplicate manual-review claim for ${entry.name} (also in ${previous})`);
else manualClaims.set(key, name);
}
const problem = manualReviewProblem(entry, ROOT);
if (problem) manualProblems.push(`${name}: attempt ${index + 1}: ${problem}`);
}
collectors.push(parsed);
const counts = collectorOutcomeCounts([parsed]);
files.push({ file: name, tier: parsed.tier ?? 'unknown', shard: parsed.shard ?? '-',
cost: parsed.total_cost_usd ?? 0, flaky: parsed.flaky_retries?.length ?? 0,
total: counts.passed + counts.failed + counts.manual_accepted, ...counts });
for (const f of parsed.flaky_retries ?? []) flaky.push({ ...f, file: name });
} catch { /* non-eval JSON — not this report's business */ }
} catch (error) {
manualProblems.push(`${name}: malformed collector JSON (${error instanceof Error ? error.message : String(error)})`);
}
}
const evidence = collectorOutcomeCounts(collectors);
console.log(`[test:paid] collector final outcomes: ${evidence.executed} executed, ${evidence.reused} reused; ${evidence.passed} passed, ${evidence.failed} failed (${evidence.attempts} attempt records from ${collectors.length} collectors)`);
console.log(`[test:paid] collector final outcomes: ${evidence.executed} executed, ${evidence.reused} reused; ${evidence.passed} passed, ${evidence.failed} failed, ${evidence.manual_accepted} manual accepted (unscored; no score-cache credit) (${evidence.attempts} attempt records from ${collectors.length} collectors)`);
if (flaky.length > 0) {
console.log(`[test:paid] report: ⚠ ${flaky.length} cases with multiple attempts this run:`);
for (const f of flaky) console.log(` ⚠ ${f.name} (x${f.attempts}) — ${f.file}`);
@@ -1525,12 +1567,24 @@ async function main(): Promise<number> {
console.log(` ⚠ ${outcome.files.join(' ')} (${outcome.executedTests} skipped — external service missing or tier mismatch)`);
}
}
if (!verdict.ok) {
if (manualProblems.length) verdict.problems.push(...manualProblems);
if (evidence.failed > 0) verdict.problems.push(`${evidence.failed} unapproved final collector failure(s)`);
if (files.reduce((sum, file) => sum + file.total, 0) !== evidence.passed + evidence.failed + evidence.manual_accepted
|| files.reduce((sum, file) => sum + file.executed + file.reused, 0) !== evidence.executed + evidence.reused) {
verdict.problems.push('Collector summary totals are inconsistent');
}
if (!manualProblems.length) fs.writeFileSync(summaryPath, JSON.stringify({ version: 1, files, totals: {
...evidence, total: evidence.passed + evidence.failed + evidence.manual_accepted,
flaky: files.reduce((sum, file) => sum + file.flaky, 0),
} }, null, 2) + '\n');
if (verdict.problems.length) {
console.error(`[test:paid] report: ${verdict.problems.length} problem(s):`);
for (const problem of verdict.problems) console.error(` ✗ ${problem}`);
return 1;
}
console.log('[test:paid] report: every planned shard accounted and passed');
console.log(evidence.manual_accepted
? `[test:paid] report: every planned shard accounted; ${evidence.manual_accepted} manual acceptance(s), no automated-score credit`
: '[test:paid] report: every planned shard accounted and passed');
return 0;
}