v1.90.0.0 feat: make browser cookie imports explicit and safe (#2964)

* fix(browse): prepare reliable cookie import wave for validation

* ci: sequence quality and behavior for validation branch

* fix(browse): isolate Windows qualification and preserve native diagnostics

* test(browse): cover cookie workflow quality and isolate Windows user paths

* test(browse): trace native member startup and initialize fresh folders

* fix(browse): keep Windows member stdin alive through EOF

* fix(browse): latch native timeouts and compare contained Edge startup

* test(browse): verify native version metadata and actual Windows argv

* test(browse): qualify Dia import on isolated macOS CI

* fix(browse): require picker origin for session mutations

* fix(browse): bound credential reads through stream completion

* test(browse): inspect owned Windows process arguments natively

* test(evals): preserve passing coverage during cookie repair reruns

* test(browse): isolate Dia qualification in a fresh macOS account

* test(browse): pass bounded integer timeouts to native Mac probes

* test(browse): distinguish Windows profile initialization from containment

* test(browse): await descendant pipe readiness before parent exit

* test(browse): initialize and restore isolated macOS Keychain state

* test(browse): initialize Windows fixture folders before qualification

* test(ci): pin the same Node runtime across Windows checks

* test(browse): distinguish native macOS browser preflight stages

* test(browse): isolate Windows descendant console lifetime

* test(browse): preserve native receipts and identify fixture lock holders

* test(browse): prepare dependency resolution before native Mac worker startup

* test(ci): include lock and close checks in native diagnostics

* test(browse): preserve native owner probe stages and subprocess deadlines

* fix(browse): classify Chromium profile-in-use exit precisely

* test(browse): retain Mac qualification evidence through cleanup failures

* test(browse): bound Mac fixture paths and retire its owned user domain

* test(browse): accept vanished fixture entries without weakening cleanup

* test(browse): identify probe-created macOS user domains safely

* test(browse): observe Mac user domains without targeting them first

* test(browse): use passive fresh-user ownership throughout Mac qualification

* test(browse): distinguish profile and registered-home Keychain lookups

* test(browse): qualify Dia under one registered account home

* test(browse): identify Dia startup and owned process-group failures

* test(browse): classify bounded Dia startup diagnostics without leaking output

* fix(test): preserve native Mac sandboxing and reap owned browser children

* fix(browse): preserve Chromium sandboxing for native profile imports

* test(browse): inspect signed Mach-O architecture without launching Xcode tools

* test(browse): sample pending Dia startup and reap on all cleanup paths

* test(browse): compare protected Dia launches in fresh Bun and Node accounts

* test(browse): inspect isolated Mac GUI readiness without browser access

* v1.90.0.0 fix: bind cookie picker actions to their document

* test: validate cookie guards and fit nested launch fixtures

* ci: configure the bundled Chromium sandbox helper

* fix(browse): classify Playwright authentication timeouts

* test: retain bounded Windows lifecycle diagnostics

* test(cso): reuse bounded NTFS precision candidates

* test(review): handle explicit preservation choices safely

* test(browse): remove owned fixture directories with explicit primitives

* test(review): distinguish descriptive reuse from edit commitments

* test: admit only the approved unscored cookie workflow refusal

* test: keep the Office Hours judge mock export-complete

* fix: keep dependency-free CI planners independent of the model SDK

* test: observe the exact holder after a native fixture unlink failure

* fix: start seeded PTY observations at owned readiness

* test: acquire identity-bound Windows deletion admission before profile resets

* test: preserve qualified Git index bits without authorizing mutations
This commit is contained in:
Garry Tan
2026-09-25 12:06:45 -04:00
committed by GitHub
parent 730a1017d1
commit a84b0b5b6d
111 changed files with 14996 additions and 1057 deletions
+15 -9
View File
@@ -22,6 +22,7 @@
import * as fs from 'node:fs';
import * as path from 'node:path';
import { getProjectEvalDir, isPartialEval, isFinalizedEvalResultFile, type EvalResult } from '../test/helpers/eval-store';
import { evalEntryOutcome } from '../test/helpers/eval-store';
import { flakeLedgerPath, type FlakeLedgerEntry } from './test-free-shards';
interface TestSeries {
@@ -29,6 +30,7 @@ interface TestSeries {
runs: number;
passes: number;
fails: number;
manualAccepted: number;
retriedPasses: number;
totalAttempts: number;
totalCostUsd: number;
@@ -54,14 +56,18 @@ export function aggregate(evalFiles: string[]): Map<string, TestSeries> {
}
for (const [name, entries] of byName) {
const s = series.get(name) ?? {
name, runs: 0, passes: 0, fails: 0, retriedPasses: 0,
name, runs: 0, passes: 0, fails: 0, manualAccepted: 0, retriedPasses: 0,
totalAttempts: 0, totalCostUsd: 0, totalDurationMs: 0, lastSeen: '',
};
const final = entries[entries.length - 1];
s.runs += 1;
s.totalAttempts += entries.length;
if (final.passed) s.passes += 1; else s.fails += 1;
if (final.passed && entries.length > 1) s.retriedPasses += 1;
const outcome = evalEntryOutcome(final);
if (outcome === 'manual-review') s.manualAccepted += 1;
else {
s.runs += 1;
if (outcome === 'passed') s.passes += 1; else s.fails += 1;
if (outcome === 'passed' && entries.length > 1) s.retriedPasses += 1;
}
for (const e of entries) {
s.totalCostUsd += e.cost_usd || 0;
s.totalDurationMs += e.duration_ms || 0;
@@ -115,21 +121,21 @@ if (import.meta.main) {
const files = collectEvalFiles(dir, sinceDays);
const series = [...aggregate(files).values()]
.sort((a, b) => b.retriedPasses - a.retriedPasses || (b.fails / b.runs) - (a.fails / a.runs));
.sort((a, b) => b.retriedPasses - a.retriedPasses || (b.fails / Math.max(1, b.runs)) - (a.fails / Math.max(1, a.runs)));
const ledger = readFreeLedger();
if (asJson) {
console.log(JSON.stringify({ dir, runsScanned: files.length, tests: series, freeLedger: ledger }, null, 2));
} else {
console.log(`flake-rank: ${files.length} finalized run file(s) under ${dir}`);
const flaky = series.filter((s) => s.retriedPasses > 0 || s.fails > 0);
const flaky = series.filter((s) => s.retriedPasses > 0 || s.fails > 0 || s.manualAccepted > 0);
if (flaky.length === 0) {
console.log(' no retried passes and no failures recorded — clean series');
} else {
console.log(' retries fails/runs avg-dur test');
console.log(' retries fails/runs manual avg-dur test');
for (const s of flaky.slice(0, 30)) {
console.log(` ${String(s.retriedPasses).padStart(7)} ${String(s.fails).padStart(5)}/${String(s.runs).padEnd(4)} `
+ `${Math.round(s.totalDurationMs / s.totalAttempts / 1000).toString().padStart(5)}s ${s.name}`);
console.log(` ${String(s.retriedPasses).padStart(7)} ${String(s.fails).padStart(5)}/${String(s.runs).padEnd(6)} `
+ `${String(s.manualAccepted).padStart(6)} ${Math.round(s.totalDurationMs / s.totalAttempts / 1000).toString().padStart(5)}s ${s.name}`);
}
}
if (ledger.length > 0) {
+11 -3
View File
@@ -7,7 +7,7 @@
*/
import * as fs from 'fs';
import { getProjectEvalDir, listEvalJsonFiles } from '../test/helpers/eval-store';
import { evalEntryOutcome, getProjectEvalDir, listEvalJsonFiles } from '../test/helpers/eval-store';
const EVAL_DIR = getProjectEvalDir();
@@ -52,6 +52,7 @@ interface RunSummary {
tier: string;
version: string;
passed: number;
manual: Array<{ name: string; approvedBy: string; approvalUrl: string }>;
total: number;
cost: number;
duration: number;
@@ -65,13 +66,19 @@ for (const file of files) {
if (filterBranch && data.branch !== filterBranch) continue;
if (filterTier && data.tier !== filterTier) continue;
const totalTurns = (data.tests || []).reduce((s: number, t: any) => s + (t.turns_used || 0), 0);
const tests = Array.isArray(data.tests) ? data.tests : null;
const final = tests ? [...new Map<string, any>(tests.map((t: any) => [t.name, t] as const)).values()] : [];
const manual = final.filter((t: any) => evalEntryOutcome(t) === 'manual-review').map((t: any) => ({
name: t.name, approvedBy: t.manual_review.approval.approved_by, approvalUrl: t.manual_review.approval.approval_url,
}));
runs.push({
file,
timestamp: data.timestamp || '',
branch: data.branch || 'unknown',
tier: data.tier || 'unknown',
version: data.version || '?',
passed: data.passed || 0,
passed: tests ? tests.filter((t: any) => evalEntryOutcome(t) === 'passed').length : data.passed || 0,
manual,
total: data.total_tests || 0,
cost: data.total_cost_usd || 0,
duration: data.total_duration_ms || 0,
@@ -110,7 +117,8 @@ for (const run of displayed) {
const cost = `$${run.cost.toFixed(2)}`.padEnd(8);
const turns = run.turns > 0 ? `${run.turns}t`.padEnd(7) : ''.padEnd(7);
const dur = run.duration > 0 ? `${Math.round(run.duration / 1000)}s`.padEnd(10) : ''.padEnd(10);
console.log(` ${date.padEnd(17)}${branch}${run.tier.padEnd(12)}${pass}${cost}${turns}${dur}v${run.version}`);
const manual = run.manual.map(entry => `MANUAL/unscored ${entry.name}: approved by ${entry.approvedBy} (${entry.approvalUrl})`).join('; ');
console.log(` ${date.padEnd(17)}${branch}${run.tier.padEnd(12)}${pass}${cost}${turns}${dur}v${run.version}${manual ? ` ${manual}` : ''}`);
}
console.log('─'.repeat(105));
+16 -2
View File
@@ -8,7 +8,7 @@
import * as fs from 'fs';
import type { EvalResult } from '../test/helpers/eval-store';
import { getProjectEvalDir, listEvalJsonFiles } from '../test/helpers/eval-store';
import { evalEntryOutcome, getProjectEvalDir, listEvalJsonFiles } from '../test/helpers/eval-store';
const EVAL_DIR = getProjectEvalDir();
@@ -77,11 +77,21 @@ const avgDetection = detectionRates.length > 0
// Flaky tests (passed in some runs, failed in others)
const testResults = new Map<string, boolean[]>();
const manualAccepted: Array<{ name: string; approvedBy: string; approvalUrl: string }> = [];
for (const r of results) {
const final = new Map(r.tests.map(t => [t.name, t]));
const manuallyAcceptedNames = new Set([...final.values()].filter(t => evalEntryOutcome(t) === 'manual-review').map(t => t.name));
for (const t of final.values()) {
if (evalEntryOutcome(t) === 'manual-review') manualAccepted.push({ name: t.name,
approvedBy: t.manual_review!.approval.approved_by, approvalUrl: t.manual_review!.approval.approval_url });
}
for (const t of r.tests) {
if (manuallyAcceptedNames.has(t.name)) continue;
const key = `${r.tier}:${t.name}`;
const outcome = evalEntryOutcome(t);
if (outcome === 'manual-review') continue;
if (!testResults.has(key)) testResults.set(key, []);
testResults.get(key)!.push(t.passed);
testResults.get(key)!.push(outcome === 'passed');
}
}
const flakyTests: string[] = [];
@@ -119,6 +129,10 @@ console.log('Eval Summary');
console.log('═'.repeat(70));
console.log(` Total runs: ${results.length} (${e2eRuns.length} e2e, ${judgeRuns.length} llm-judge)`);
console.log(` Total spend: $${totalCost.toFixed(2)}`);
if (manualAccepted.length) {
console.log(` Manual accepted: ${manualAccepted.length} unscored provider refusal(s)`);
for (const entry of manualAccepted) console.log(` ${entry.name}: approved by ${entry.approvedBy} (${entry.approvalUrl})`);
}
console.log(` Avg cost/e2e: $${avgE2ECost.toFixed(2)}`);
console.log(` Avg cost/judge: $${avgJudgeCost.toFixed(2)}`);
if (avgE2EDuration > 0) {
+9 -13
View File
@@ -11,7 +11,8 @@
import * as fs from 'fs';
import * as path from 'path';
import * as os from 'os';
import { getProjectEvalDir } from '../test/helpers/eval-store';
import { evalEntryOutcome, getProjectEvalDir } from '../test/helpers/eval-store';
import type { EvalTestEntry } from '../test/helpers/eval-store';
const GSTACK_DEV_DIR = path.join(os.homedir(), '.gstack-dev');
// Heartbeat + per-run progress logs are GLOBAL by design — session-runner.ts
@@ -38,16 +39,7 @@ export interface HeartbeatData {
}
export interface PartialData {
tests: Array<{
name: string;
suite?: string;
attempt?: number;
passed: boolean;
cost_usd: number;
duration_ms: number;
turns_used?: number;
exit_reason?: string;
}>;
tests: Array<Partial<EvalTestEntry> & Pick<EvalTestEntry, 'name' | 'passed' | 'cost_usd' | 'duration_ms'>>;
total_cost_usd: number;
_partial?: boolean;
}
@@ -120,12 +112,14 @@ export function renderDashboard(heartbeat: HeartbeatData | null, partial: Partia
// Completed tests from partial
if (partial?.tests) {
for (const t of partial.tests) {
const icon = t.passed ? '\u2713' : '\u2717';
const manual = evalEntryOutcome(t) === 'manual-review';
const icon = manual ? 'M' : evalEntryOutcome(t) === 'passed' ? '\u2713' : '\u2717';
const cost = `$${t.cost_usd.toFixed(2)}`;
const dur = `${Math.round(t.duration_ms / 1000)}s`;
const turns = t.turns_used !== undefined ? `${t.turns_used} turns` : '';
const name = t.name.length > 30 ? t.name.slice(0, 27) + '...' : t.name.padEnd(30);
lines.push(` ${icon} ${name} ${cost.padStart(6)} ${dur.padStart(5)} ${turns}`);
const approval = manual ? ` MANUAL/unscored; approved by ${t.manual_review!.approval.approved_by} (${t.manual_review!.approval.approval_url})` : '';
lines.push(` ${icon} ${name} ${cost.padStart(6)} ${dur.padStart(5)} ${turns}${approval}`);
}
}
@@ -151,6 +145,8 @@ export function renderDashboard(heartbeat: HeartbeatData | null, partial: Partia
const totalCost = partial?.total_cost_usd || 0;
const running = heartbeat?.status === 'running' ? 1 : 0;
lines.push(` Completed: ${completedCount} Running: ${running} Cost: $${totalCost.toFixed(2)} Elapsed: ${formatDuration(elapsed)}`);
const manualAccepted = partial?.tests?.filter(t => evalEntryOutcome(t) === 'manual-review').length ?? 0;
if (manualAccepted) lines.push(` Manual accepted: ${manualAccepted} unscored provider refusal(s)`);
if (heartbeat?.runId) {
const logPath = path.join(GSTACK_DEV_DIR, 'e2e-runs', heartbeat.runId, 'progress.log');
+65 -11
View File
@@ -66,7 +66,8 @@ import {
import { PAID_TEST_GLOBS, isPaidTestFile } from '../test/helpers/paid-test-set';
import { PERIODIC_CI_EXCLUDE } from '../test/helpers/periodic-exclude-data';
import { AUTOPLAN_CHAIN_BUDGET, FILE_RETRY_BUDGETS, STRICT_RETRY_CASE_BUDGETS } from '../test/helpers/eval-budgets';
import { getProjectEvalDir, getClaudeCliVersion, isFinalizedEvalResultFile } from '../test/helpers/eval-store';
import { getProjectEvalDir, getClaudeCliVersion, isFinalizedEvalResultFile, evalEntryOutcome } from '../test/helpers/eval-store';
import { manualReviewProblem } from '../test/helpers/cookie-workflow-manual-review';
import { preflightAnthropicApi } from '../test/helpers/anthropic-preflight';
import { OVERLAY_MIN_FILE_WALL_MS } from '../test/helpers/overlay-case-policy';
import { PR_PROFILE_CASE_IDS, PR_PROFILE_FILES, packageChangeOnlyVersion, selectPrProfile, type PrProfileSelection } from './test-pr-profile';
@@ -1314,19 +1315,20 @@ export function formatProfileCoverage(manifest: PaidRunManifest): string[] {
/** Final outcomes use each case's last attempt; the attempt total stays visible. */
export function collectorOutcomeCounts(results: Array<{ tests?: Array<{
name: string; suite?: string; passed: boolean; execution?: string;
}> }>): { executed: number; reused: number; passed: number; failed: number; attempts: number } {
const counts = { executed: 0, reused: 0, passed: 0, failed: 0, attempts: 0 };
name: string; suite?: string; passed: boolean; execution?: string; manual_review?: unknown;
}> }>): { executed: number; reused: number; passed: number; failed: number; manual_accepted: number; attempts: number } {
const counts = { executed: 0, reused: 0, passed: 0, failed: 0, manual_accepted: 0, attempts: 0 };
for (const result of results) {
const cases = new Map<string, NonNullable<typeof result.tests>[number]>();
for (const entry of result.tests ?? []) {
if (typeof entry.name !== 'string' || typeof entry.passed !== 'boolean') continue;
if (!entry || typeof entry !== 'object' || typeof entry.name !== 'string' || typeof entry.passed !== 'boolean') continue;
counts.attempts++;
cases.set(`${entry.suite ?? ''}\0${entry.name}`, entry);
}
for (const entry of cases.values()) {
counts[entry.execution === 'reused' ? 'reused' : 'executed']++;
counts[entry.passed ? 'passed' : 'failed']++;
const outcome = evalEntryOutcome(entry);
counts[outcome === 'manual-review' ? 'manual_accepted' : outcome]++;
}
}
return counts;
@@ -1479,6 +1481,8 @@ async function main(): Promise<number> {
// ── Report mode: reconcile slice artifacts against the manifest. Fail-closed:
// a slice whose artifact never landed is a FAILURE, not an absence.
if (options.reportDir) {
const summaryPath = path.join(options.reportDir, 'collector-outcomes.json');
fs.rmSync(summaryPath, { force: true });
const manifest = parseRunManifest(fs.readFileSync(path.join(options.reportDir, 'manifest.json'), 'utf-8'));
const results: SliceResult[] = fs.readdirSync(options.reportDir)
.filter((name) => /^slice-\d+\.json$/.test(name))
@@ -1498,16 +1502,54 @@ async function main(): Promise<number> {
// Source: the finalized eval-store JSONs inside the slice artifacts.
const flaky: Array<{ name: string; attempts: number; file: string }> = [];
const collectors: Parameters<typeof collectorOutcomeCounts>[0] = [];
const files: Array<{ file: string; tier: string; shard: string | number; cost: number;
flaky: number; total: number; executed: number; reused: number; passed: number;
failed: number; manual_accepted: number; attempts: number }> = [];
const manualProblems: string[] = [];
const manualClaims = new Map<string, string>();
for (const name of fs.readdirSync(options.reportDir, { recursive: true }) as string[]) {
if (!isFinalizedEvalResultFile(name)) continue;
try {
const parsed = JSON.parse(fs.readFileSync(path.join(options.reportDir, name), 'utf-8'));
if (Array.isArray(parsed.tests)) collectors.push(parsed);
if (!Array.isArray(parsed.tests)) {
if (Object.hasOwn(parsed, 'tests') || parsed.total_tests !== undefined || parsed.manual_review !== undefined) {
manualProblems.push(`${name}: malformed collector tests[]`);
}
continue;
}
const seen = new Map<string, number>();
for (const [index, entry] of parsed.tests.entries()) {
if (!entry || typeof entry !== 'object' || typeof entry.name !== 'string' || !entry.name
|| typeof entry.passed !== 'boolean') {
manualProblems.push(`${name}: attempt ${index + 1}: malformed collector entry (name/passed required)`);
continue;
}
const key = `${entry.suite ?? ''}\0${entry.name}`;
const occurrence = (seen.get(key) ?? 0) + 1;
seen.set(key, occurrence);
if (Object.hasOwn(entry, 'manual_review') && occurrence !== 1) {
manualProblems.push(`${name}: attempt ${index + 1}: manual review is only valid on the first case attempt`);
}
if (Object.hasOwn(entry, 'manual_review')) {
const previous = manualClaims.get(key);
if (previous && previous !== name) manualProblems.push(`${name}: duplicate manual-review claim for ${entry.name} (also in ${previous})`);
else manualClaims.set(key, name);
}
const problem = manualReviewProblem(entry, ROOT);
if (problem) manualProblems.push(`${name}: attempt ${index + 1}: ${problem}`);
}
collectors.push(parsed);
const counts = collectorOutcomeCounts([parsed]);
files.push({ file: name, tier: parsed.tier ?? 'unknown', shard: parsed.shard ?? '-',
cost: parsed.total_cost_usd ?? 0, flaky: parsed.flaky_retries?.length ?? 0,
total: counts.passed + counts.failed + counts.manual_accepted, ...counts });
for (const f of parsed.flaky_retries ?? []) flaky.push({ ...f, file: name });
} catch { /* non-eval JSON — not this report's business */ }
} catch (error) {
manualProblems.push(`${name}: malformed collector JSON (${error instanceof Error ? error.message : String(error)})`);
}
}
const evidence = collectorOutcomeCounts(collectors);
console.log(`[test:paid] collector final outcomes: ${evidence.executed} executed, ${evidence.reused} reused; ${evidence.passed} passed, ${evidence.failed} failed (${evidence.attempts} attempt records from ${collectors.length} collectors)`);
console.log(`[test:paid] collector final outcomes: ${evidence.executed} executed, ${evidence.reused} reused; ${evidence.passed} passed, ${evidence.failed} failed, ${evidence.manual_accepted} manual accepted (unscored; no score-cache credit) (${evidence.attempts} attempt records from ${collectors.length} collectors)`);
if (flaky.length > 0) {
console.log(`[test:paid] report: ⚠ ${flaky.length} cases with multiple attempts this run:`);
for (const f of flaky) console.log(` ⚠ ${f.name} (x${f.attempts}) — ${f.file}`);
@@ -1525,12 +1567,24 @@ async function main(): Promise<number> {
console.log(` ⚠ ${outcome.files.join(' ')} (${outcome.executedTests} skipped — external service missing or tier mismatch)`);
}
}
if (!verdict.ok) {
if (manualProblems.length) verdict.problems.push(...manualProblems);
if (evidence.failed > 0) verdict.problems.push(`${evidence.failed} unapproved final collector failure(s)`);
if (files.reduce((sum, file) => sum + file.total, 0) !== evidence.passed + evidence.failed + evidence.manual_accepted
|| files.reduce((sum, file) => sum + file.executed + file.reused, 0) !== evidence.executed + evidence.reused) {
verdict.problems.push('Collector summary totals are inconsistent');
}
if (!manualProblems.length) fs.writeFileSync(summaryPath, JSON.stringify({ version: 1, files, totals: {
...evidence, total: evidence.passed + evidence.failed + evidence.manual_accepted,
flaky: files.reduce((sum, file) => sum + file.flaky, 0),
} }, null, 2) + '\n');
if (verdict.problems.length) {
console.error(`[test:paid] report: ${verdict.problems.length} problem(s):`);
for (const problem of verdict.problems) console.error(` ✗ ${problem}`);
return 1;
}
console.log('[test:paid] report: every planned shard accounted and passed');
console.log(evidence.manual_accepted
? `[test:paid] report: every planned shard accounted; ${evidence.manual_accepted} manual acceptance(s), no automated-score credit`
: '[test:paid] report: every planned shard accounted and passed');
return 0;
}