Files
gstack/scripts/eval-flake-rank.ts
T
Garry Tan a84b0b5b6d v1.90.0.0 feat: make browser cookie imports explicit and safe (#2964)
* fix(browse): prepare reliable cookie import wave for validation

* ci: sequence quality and behavior for validation branch

* fix(browse): isolate Windows qualification and preserve native diagnostics

* test(browse): cover cookie workflow quality and isolate Windows user paths

* test(browse): trace native member startup and initialize fresh folders

* fix(browse): keep Windows member stdin alive through EOF

* fix(browse): latch native timeouts and compare contained Edge startup

* test(browse): verify native version metadata and actual Windows argv

* test(browse): qualify Dia import on isolated macOS CI

* fix(browse): require picker origin for session mutations

* fix(browse): bound credential reads through stream completion

* test(browse): inspect owned Windows process arguments natively

* test(evals): preserve passing coverage during cookie repair reruns

* test(browse): isolate Dia qualification in a fresh macOS account

* test(browse): pass bounded integer timeouts to native Mac probes

* test(browse): distinguish Windows profile initialization from containment

* test(browse): await descendant pipe readiness before parent exit

* test(browse): initialize and restore isolated macOS Keychain state

* test(browse): initialize Windows fixture folders before qualification

* test(ci): pin the same Node runtime across Windows checks

* test(browse): distinguish native macOS browser preflight stages

* test(browse): isolate Windows descendant console lifetime

* test(browse): preserve native receipts and identify fixture lock holders

* test(browse): prepare dependency resolution before native Mac worker startup

* test(ci): include lock and close checks in native diagnostics

* test(browse): preserve native owner probe stages and subprocess deadlines

* fix(browse): classify Chromium profile-in-use exit precisely

* test(browse): retain Mac qualification evidence through cleanup failures

* test(browse): bound Mac fixture paths and retire its owned user domain

* test(browse): accept vanished fixture entries without weakening cleanup

* test(browse): identify probe-created macOS user domains safely

* test(browse): observe Mac user domains without targeting them first

* test(browse): use passive fresh-user ownership throughout Mac qualification

* test(browse): distinguish profile and registered-home Keychain lookups

* test(browse): qualify Dia under one registered account home

* test(browse): identify Dia startup and owned process-group failures

* test(browse): classify bounded Dia startup diagnostics without leaking output

* fix(test): preserve native Mac sandboxing and reap owned browser children

* fix(browse): preserve Chromium sandboxing for native profile imports

* test(browse): inspect signed Mach-O architecture without launching Xcode tools

* test(browse): sample pending Dia startup and reap on all cleanup paths

* test(browse): compare protected Dia launches in fresh Bun and Node accounts

* test(browse): inspect isolated Mac GUI readiness without browser access

* v1.90.0.0 fix: bind cookie picker actions to their document

* test: validate cookie guards and fit nested launch fixtures

* ci: configure the bundled Chromium sandbox helper

* fix(browse): classify Playwright authentication timeouts

* test: retain bounded Windows lifecycle diagnostics

* test(cso): reuse bounded NTFS precision candidates

* test(review): handle explicit preservation choices safely

* test(browse): remove owned fixture directories with explicit primitives

* test(review): distinguish descriptive reuse from edit commitments

* test: admit only the approved unscored cookie workflow refusal

* test: keep the Office Hours judge mock export-complete

* fix: keep dependency-free CI planners independent of the model SDK

* test: observe the exact holder after a native fixture unlink failure

* fix: start seeded PTY observations at owned readiness

* test: acquire identity-bound Windows deletion admission before profile resets

* test: preserve qualified Git index bits without authorizing mutations
2026-09-25 12:06:45 -04:00

151 lines
5.9 KiB
TypeScript

#!/usr/bin/env bun
/**
* eval-flake-rank — the flake-telemetry dial (WS1).
*
* Aggregates per-test series across every FINALIZED eval-store run on this
* machine (default: ~/.gstack/projects/<slug>/evals/, shard dirs included)
* plus the free suite's flake ledger, and ranks tests by flake signal:
* retried passes first (a test that needs attempt 2 to go green is the
* definition of a flake), then failure rate.
*
* This is the readable dial behind two policies:
* - a flaky pass never blocks a merge, but it is recorded and RANKED here;
* - the required-check promotion (WS16) needs weeks of clean flake-rank,
* not vibes.
*
* Usage:
* bun run eval:flake-rank # project eval dir
* bun run eval:flake-rank --dir <path> # e.g. downloaded CI artifacts
* bun run eval:flake-rank --json # machine-readable
*/
import * as fs from 'node:fs';
import * as path from 'node:path';
import { getProjectEvalDir, isPartialEval, isFinalizedEvalResultFile, type EvalResult } from '../test/helpers/eval-store';
import { evalEntryOutcome } from '../test/helpers/eval-store';
import { flakeLedgerPath, type FlakeLedgerEntry } from './test-free-shards';
interface TestSeries {
name: string;
runs: number;
passes: number;
fails: number;
manualAccepted: number;
retriedPasses: number;
totalAttempts: number;
totalCostUsd: number;
totalDurationMs: number;
lastSeen: string;
}
export function aggregate(evalFiles: string[]): Map<string, TestSeries> {
const series = new Map<string, TestSeries>();
for (const file of evalFiles) {
let run: EvalResult;
try {
run = JSON.parse(fs.readFileSync(file, 'utf-8'));
} catch { continue; }
if (isPartialEval(run, file)) continue; // in-progress accumulators are not runs
if (!Array.isArray(run.tests)) continue;
// Group this run's entries by name so N attempts = 1 run of that test.
const byName = new Map<string, typeof run.tests>();
for (const t of run.tests) {
const list = byName.get(t.name) ?? [];
list.push(t);
byName.set(t.name, list);
}
for (const [name, entries] of byName) {
const s = series.get(name) ?? {
name, runs: 0, passes: 0, fails: 0, manualAccepted: 0, retriedPasses: 0,
totalAttempts: 0, totalCostUsd: 0, totalDurationMs: 0, lastSeen: '',
};
const final = entries[entries.length - 1];
s.totalAttempts += entries.length;
const outcome = evalEntryOutcome(final);
if (outcome === 'manual-review') s.manualAccepted += 1;
else {
s.runs += 1;
if (outcome === 'passed') s.passes += 1; else s.fails += 1;
if (outcome === 'passed' && entries.length > 1) s.retriedPasses += 1;
}
for (const e of entries) {
s.totalCostUsd += e.cost_usd || 0;
s.totalDurationMs += e.duration_ms || 0;
}
if (run.timestamp > s.lastSeen) s.lastSeen = run.timestamp;
series.set(name, s);
}
}
return series;
}
export function collectEvalFiles(dir: string, sinceDays = 60): string[] {
if (!fs.existsSync(dir)) return [];
const cutoff = Date.now() - sinceDays * 86_400_000;
const out: string[] = [];
for (const name of fs.readdirSync(dir, { recursive: true }) as string[]) {
if (!isFinalizedEvalResultFile(name)) continue;
const full = path.join(dir, name);
try {
// Recency bound (review finding): E2E results embed full transcripts
// (MBs each) and the scan is otherwise unbounded over all-time history.
if (fs.statSync(full).mtimeMs < cutoff) continue;
} catch { continue; }
out.push(full);
}
return out;
}
function readFreeLedger(): FlakeLedgerEntry[] {
// Per-LINE parse: one malformed JSONL line (torn write, manual edit) must
// drop that line, never vanish the whole series (codex adversarial finding).
let raw: string;
try {
raw = fs.readFileSync(flakeLedgerPath(), 'utf-8');
} catch { return []; }
const out: FlakeLedgerEntry[] = [];
for (const line of raw.split('\n')) {
if (!line.trim()) continue;
try { out.push(JSON.parse(line)); } catch { /* torn line — skip */ }
}
return out;
}
if (import.meta.main) {
const argv = process.argv.slice(2);
const dirFlag = argv.indexOf('--dir');
const dir = dirFlag !== -1 ? argv[dirFlag + 1] : getProjectEvalDir();
const asJson = argv.includes('--json');
const sinceFlag = argv.indexOf('--since-days');
const sinceDays = sinceFlag !== -1 ? Number(argv[sinceFlag + 1]) || 60 : 60;
const files = collectEvalFiles(dir, sinceDays);
const series = [...aggregate(files).values()]
.sort((a, b) => b.retriedPasses - a.retriedPasses || (b.fails / Math.max(1, b.runs)) - (a.fails / Math.max(1, a.runs)));
const ledger = readFreeLedger();
if (asJson) {
console.log(JSON.stringify({ dir, runsScanned: files.length, tests: series, freeLedger: ledger }, null, 2));
} else {
console.log(`flake-rank: ${files.length} finalized run file(s) under ${dir}`);
const flaky = series.filter((s) => s.retriedPasses > 0 || s.fails > 0 || s.manualAccepted > 0);
if (flaky.length === 0) {
console.log(' no retried passes and no failures recorded — clean series');
} else {
console.log(' retries fails/runs manual avg-dur test');
for (const s of flaky.slice(0, 30)) {
console.log(` ${String(s.retriedPasses).padStart(7)} ${String(s.fails).padStart(5)}/${String(s.runs).padEnd(6)} `
+ `${String(s.manualAccepted).padStart(6)} ${Math.round(s.totalDurationMs / s.totalAttempts / 1000).toString().padStart(5)}s ${s.name}`);
}
}
if (ledger.length > 0) {
const byFile = new Map<string, number>();
for (const e of ledger) byFile.set(e.file, (byFile.get(e.file) ?? 0) + 1);
console.log(`free-suite flaky-passes (${flakeLedgerPath()}):`);
for (const [file, n] of [...byFile.entries()].sort((a, b) => b[1] - a[1])) {
console.log(` ${String(n).padStart(3)}x ${file}`);
}
}
}
}