#!/usr/bin/env bun /** * eval-pass-rates (alias: eval-flake-rank) — per-case trial pass rates. * * Reads trial records (one JSONL line per trial: case, kind, trial, outcome, * exit_reason, duration, cost, model, CLI version, series identity, run id, * sha, policy_version) from the last N completed `evals-periodic.yml` runs on * the current branch and `main` (downloading only each run's small * `trial-outcomes` artifact through `gh`), plus any local eval dirs, and * prints per-case per-trial pass rates with 95% Wilson intervals. * * A series is one case under one input identity: the case's own touchfiles * minus GLOBAL_TOUCHFILES (`caseSeriesIdentities`), grouped by model and CLI * version, per policy_version. A new identity starts a new series; earlier * series stay visible. Only post-policy trials of the current series feed the * labels and alarms. Legacy eval-store records (`--backfill`, `--dir`) are * imported as pre-policy trials (first attempt only; a missing attempt means * 1) and are display-only. * * Labels: INCONCLUSIVE (below the entry rule's minimum trials), BROKEN (latest run 0/n * after a prior interval at or above the entry rate), FLAKY (failures and an * interval straddling the entry rate), FAILING (interval below the entry * rate), PASSING (otherwise). * * The weekly gate (`--gate`) exits non-zero with ACTION REQUIRED when a * non-quarantined case meets the quarantine entry rule, a rule case behaves * like a behavior case, a blocking case's current-identity rate is * significantly below its previous identity (one-sided Fisher exact, * Holm-controlled across cases), or a CASE_QUARANTINE entry has met its exit * rule, expired, or pushed its tier over the cap. History that cannot be * fetched fails the gate closed. * * Usage: * bun run eval:pass-rates # last 10 weekly runs, this branch + main * bun run eval:pass-rates --case --runs 20 * bun run eval:pass-rates --dir # local eval dirs / downloaded artifacts (repeatable) * bun run eval:pass-rates --backfill # also import legacy slice artifacts, labeled pre-policy * bun run eval:pass-rates --json | --gate */ import * as fs from 'node:fs'; import * as os from 'node:os'; import * as path from 'node:path'; import { spawnSync } from 'node:child_process'; import { createHash } from 'node:crypto'; import { isPartialEval, isFinalizedEvalResultFile, evalEntryOutcome, failureClassOf, parseTrialOutcomes, sanitizeTrialError, TRIAL_OUTCOME_SCHEMA, type EvalCaseKind, type EvalResult, type TrialOutcomeRecord } from '../test/helpers/eval-store'; import { flakeLedgerPath, type FlakeLedgerEntry } from './test-free-shards'; import { E2E_KINDS, E2E_TIERS, E2E_TOUCHFILES, GLOBAL_TOUCHFILES, LLM_JUDGE_TOUCHFILES } from '../test/helpers/touchfiles-data'; import { CASE_QUARANTINE, EVAL_POLICY } from '../test/helpers/periodic-exclude-data'; import { matchGlob } from '../test/helpers/test-selection'; import { CASE_TEST_NAMES } from './test-paid-shards'; interface TestSeries { name: string; runs: number; passes: number; fails: number; manualAccepted: number; retriedPasses: number; totalAttempts: number; totalCostUsd: number; totalDurationMs: number; lastSeen: string; } export function aggregate(evalFiles: string[]): Map { const series = new Map(); for (const file of evalFiles) { let run: EvalResult; try { run = JSON.parse(fs.readFileSync(file, 'utf-8')); } catch { continue; } if (isPartialEval(run, file)) continue; // in-progress accumulators are not runs if (!Array.isArray(run.tests)) continue; // Group this run's entries by name so N attempts = 1 run of that test. const byName = new Map(); for (const t of run.tests) { const list = byName.get(t.name) ?? []; list.push(t); byName.set(t.name, list); } for (const [name, entries] of byName) { const s = series.get(name) ?? { name, runs: 0, passes: 0, fails: 0, manualAccepted: 0, retriedPasses: 0, totalAttempts: 0, totalCostUsd: 0, totalDurationMs: 0, lastSeen: '', }; const final = entries[entries.length - 1]; s.totalAttempts += entries.length; const outcome = evalEntryOutcome(final); if (outcome === 'manual-review') s.manualAccepted += 1; else { s.runs += 1; if (outcome === 'passed') s.passes += 1; else s.fails += 1; if (outcome === 'passed' && entries.length > 1) s.retriedPasses += 1; } for (const e of entries) { s.totalCostUsd += e.cost_usd || 0; s.totalDurationMs += e.duration_ms || 0; } if (run.timestamp > s.lastSeen) s.lastSeen = run.timestamp; series.set(name, s); } } return series; } export function collectEvalFiles(dir: string, sinceDays = 60): string[] { if (!fs.existsSync(dir)) return []; const cutoff = Date.now() - sinceDays * 86_400_000; const out: string[] = []; for (const name of fs.readdirSync(dir, { recursive: true }) as string[]) { if (!isFinalizedEvalResultFile(name)) continue; const full = path.join(dir, name); try { // Recency bound (review finding): E2E results embed full transcripts // (MBs each) and the scan is otherwise unbounded over all-time history. if (fs.statSync(full).mtimeMs < cutoff) continue; } catch { continue; } out.push(full); } return out; } function readFreeLedger(): FlakeLedgerEntry[] { // Per-LINE parse: one malformed JSONL line (torn write, manual edit) must // drop that line, never vanish the whole series (codex adversarial finding). let raw: string; try { raw = fs.readFileSync(flakeLedgerPath(), 'utf-8'); } catch { return []; } const out: FlakeLedgerEntry[] = []; for (const line of raw.split('\n')) { if (!line.trim()) continue; try { out.push(JSON.parse(line)); } catch { /* torn line — skip */ } } return out; } // --- Trial records --- /** * A trial record as pass-rates reads it: eval-store's trial-outcomes schema * plus the series identity the report job stamps (caseSeriesIdentities). * policy_version 0 marks a pre-policy (backfilled) record. */ export type TrialRecord = TrialOutcomeRecord & { series_identity?: string }; /** Per-file cap for downloaded artifacts: pass-rates parses data only, never executes it. */ export const TRIAL_OUTCOMES_MAX_BYTES = 8 * 1024 * 1024; /** Every `trial-outcomes*.jsonl` file under a directory, size-capped, schema-validated by eval-store. */ export function readTrialOutcomeDir(dir: string): { records: TrialRecord[]; errors: string[] } { const records: TrialRecord[] = []; const errors: string[] = []; if (!fs.existsSync(dir)) return { records, errors }; for (const name of fs.readdirSync(dir, { recursive: true }) as string[]) { if (!/(^|\/)trial-outcomes[^/]*\.jsonl$/.test(name)) continue; const full = path.join(dir, name); const parsed = parseTrialOutcomes(fs.readFileSync(full, 'utf8'), { maxBytes: TRIAL_OUTCOMES_MAX_BYTES }); records.push(...parsed.records.map(record => ({ ...record, series_identity: typeof (record as TrialRecord).series_identity === 'string' ? (record as TrialRecord).series_identity!.slice(0, 64) : undefined }))); errors.push(...parsed.errors.map(error => `${full}: ${error}`)); } return { records, errors }; } // --- Registry attribution and series identity --- export interface Registry { kinds: Record; tiers: Record; touchfiles: Record; judgeTouchfiles: Record; globals: readonly string[]; testNames: Record; } export const LIVE_REGISTRY: Registry = { kinds: E2E_KINDS, tiers: E2E_TIERS, touchfiles: E2E_TOUCHFILES, judgeTouchfiles: LLM_JUDGE_TOUCHFILES, globals: GLOBAL_TOUCHFILES, testNames: CASE_TEST_NAMES, }; /** A case's tier: its E2E_TIERS value, or 'judge' for an LLM-judge entry. */ export function caseTier(id: string, registry: Registry = LIVE_REGISTRY): string { return registry.tiers[id] ?? (id in registry.judgeTouchfiles ? 'judge' : 'unknown'); } /** * Attribute a legacy eval-store record to a registry id: the case-shard slug * suffix (`--`), the recorded name or its exact slug (`/qa b6-static` * is `qa-b6-static`), a CASE_TEST_NAMES label, or the only id its shard file * registers. Anything else is unattributed (null). */ export function attributeLegacyRecord(name: string, shard: string | undefined, registry: Registry = LIVE_REGISTRY): string | null { const known = (id: string) => id in registry.kinds; const [slugFile, slugCase] = (shard ?? '').split('--'); if (slugCase && known(slugCase)) return slugCase; if (known(name)) return name; const slug = name.toLowerCase().replace(/[^a-z0-9]+/g, '-').replace(/^-+|-+$/g, ''); if (known(slug)) return slug; const labeled = Object.entries(registry.testNames).find(([, label]) => label === name)?.[0]; if (labeled && known(labeled)) return labeled; if (slugFile) { const file = `test/${slugFile}.test.ts`; const owners = Object.keys(registry.touchfiles).filter(id => registry.touchfiles[id]!.includes(file)); if (owners.length === 1 && known(owners[0]!)) return owners[0]!; } return null; } /** * Series identity per case: a hash of the git blob ids of the files matching * the case's own touchfiles, excluding GLOBAL_TOUCHFILES (harness edits are * markers, not new series). The report job stamps this on every trial record. */ export function caseSeriesIdentities(ids: string[], root: string, registry: Registry = LIVE_REGISTRY): Record { const listed = spawnSync('git', ['ls-files', '-s'], { cwd: root, encoding: 'utf8', timeout: 20_000, maxBuffer: 64 * 1024 * 1024 }); if (listed.status !== 0) throw new Error(`git ls-files failed: ${listed.stderr}`); const blobs = listed.stdout.split('\n').filter(Boolean).map(line => { const [meta, file] = line.split('\t'); return { file: file!, blob: meta!.split(' ')[1]! }; }).filter(entry => !registry.globals.some(pattern => matchGlob(entry.file, pattern))); return Object.fromEntries(ids.map(id => { const patterns = registry.touchfiles[id] ?? registry.judgeTouchfiles[id] ?? []; const lines = blobs.filter(entry => patterns.some(pattern => matchGlob(entry.file, pattern))) .map(entry => `${entry.file} ${entry.blob}`).sort(); return [id, createHash('sha256').update(`${id}\n${lines.join('\n')}`).digest('hex').slice(0, 16)]; })); } /** * Import legacy eval-store result files as pre-policy trials (policy_version * 0, source 'backfill'): first attempt only (a missing attempt means 1), * attributed by registry id, never guessed. A manual-review acceptance carries * no automated verdict: it is counted and shown, never scored. Without a CI * run, each local result file is its own run. */ export function backfillEvalFiles(files: string[], run?: { run_id: string; sha?: string; timestamp?: string }, registry: Registry = LIVE_REGISTRY): { records: TrialRecord[]; unattributed: string[]; manualReviews: string[] } { const records: TrialRecord[] = []; const unattributed = new Set(); const manualReviews: string[] = []; for (const file of files) { let result: EvalResult & { shard?: string; claude_cli_version?: string }; try { result = JSON.parse(fs.readFileSync(file, 'utf8')); } catch { continue; } if (isPartialEval(result, file) || !Array.isArray(result.tests)) continue; const seen = new Set(); for (const entry of result.tests) { if ((entry.attempt ?? 1) !== 1 || seen.has(entry.name)) continue; seen.add(entry.name); const id = attributeLegacyRecord(entry.name, result.shard, registry); if (!id) { unattributed.add(entry.name); continue; } const outcome = evalEntryOutcome(entry); if (outcome === 'manual-review') { manualReviews.push(id); continue; } records.push({ schema: TRIAL_OUTCOME_SCHEMA, case: id, file: result.shard ? `test/${result.shard.split('--')[0]}.test.ts` : 'unknown', tier: caseTier(id, registry), kind: registry.kinds[id]!, trial: 1, panel: { n: 1, k: 1 }, attempt: 1, outcome, ...(outcome === 'failed' ? { failure_class: failureClassOf(entry) } : {}), exit_reason: entry.exit_reason, error: sanitizeTrialError(entry.error), duration_ms: Math.max(0, entry.duration_ms || 0), cost_usd: Math.max(0, entry.cost_usd || 0), model: entry.model, cli_version: result.claude_cli_version, policy_version: 0, quarantined: false, execution: entry.execution === 'reused' ? 'reused' : 'executed', source: 'backfill', run_id: run?.run_id ?? `local:${file}`, sha: run?.sha ?? result.git_sha, recorded_at: run?.timestamp ?? result.timestamp, }); } } return { records, unattributed: [...unattributed].sort(), manualReviews }; } // --- Statistics --- /** 95% Wilson score interval for k successes in n trials. */ export function wilsonInterval(k: number, n: number, z = 1.96): { lo: number; hi: number } { if (n <= 0) return { lo: 0, hi: 1 }; const p = k / n, z2 = z * z, denom = 1 + z2 / n; const center = (p + z2 / (2 * n)) / denom; const half = (z * Math.sqrt(p * (1 - p) / n + z2 / (4 * n * n))) / denom; return { lo: Math.max(0, center - half), hi: k === n ? 1 : Math.min(1, center + half) }; } function logChoose(n: number, k: number): number { let sum = 0; for (let i = 1; i <= k; i++) sum += Math.log(n - k + i) - Math.log(i); return sum; } /** * One-sided Fisher exact p-value that the CURRENT pass rate is below the * PREVIOUS one: P(X <= curPass) under the hypergeometric null with the * observed margins. */ export function fisherOneSidedLower(curPass: number, curN: number, prevPass: number, prevN: number): number { const passes = curPass + prevPass, total = curN + prevN; const denom = logChoose(total, passes); let p = 0; for (let x = Math.max(0, passes - prevN); x <= curPass; x++) p += Math.exp(logChoose(curN, x) + logChoose(prevN, passes - x) - denom); return Math.min(1, p); } /** Holm step-down: the indices whose p-values are rejected at family-wise alpha. */ export function holmRejections(pValues: number[], alpha: number): Set { const order = pValues.map((p, index) => ({ p, index })).sort((a, b) => a.p - b.p); const rejected = new Set(); for (let rank = 0; rank < order.length; rank++) { if (order[rank]!.p > alpha / (order.length - rank)) break; rejected.add(order[rank]!.index); } return rejected; } // --- Analysis --- export type PassRateLabel = 'INCONCLUSIVE' | 'BROKEN' | 'FLAKY' | 'FAILING' | 'PASSING'; export type AlarmKind = 'drift' | 'rule-as-behavior' | 'regression' | 'quarantine-exit' | 'quarantine-expired' | 'quarantine-cap' | 'quarantine-invalid'; /** The EVAL_POLICY fields pass-rates reads (structural, so tests can vary them). */ export interface PassRatePolicy { version: number; quarantine: { entry: { rate: number; minTrials: number }; exit: { rate: number; minTrials: number }; capFraction: number; expiryWeeklyRuns: number }; drift: { fisherAlpha: number; fisherMinPerSide: number }; } export type QuarantineEntry = (typeof CASE_QUARANTINE)[string]; /** Tiers whose cases block a lane; quarantine applies only to them. */ export const BLOCKING_TIERS: readonly string[] = ['gate', 'periodic']; const QUARANTINE_FAILURE_CLASSES: readonly string[] = ['detector', 'harness', 'model-latency']; export interface SeriesStats { key: string; identity: string; model: string; cli: string; policyVersion: number; passes: number; /** Scored trials: passed + failed (skipped trials carry no verdict). */ trials: number; infra: number; interval: { lo: number; hi: number }; firstSeen: string; lastSeen: string; runs: string[]; } export interface CasePassRate { case: string; kind: EvalCaseKind; tier: string; quarantined: boolean; label: PassRateLabel; /** Manual-review acceptances: visible, never scored. */ manualReviews: number; current: SeriesStats | null; previous: SeriesStats | null; prePolicy: SeriesStats | null; series: SeriesStats[]; latestRun: { runId: string; passes: number; trials: number } | null; } export interface Alarm { kind: AlarmKind; case: string; message: string } export interface PassRateReport { policyVersion: number; cases: CasePassRate[]; alarms: Alarm[]; postPolicyTrials: number; prePolicyTrials: number; unattributed: string[]; errors: string[]; } export interface AnalyzeOptions { registry?: Registry; quarantine?: Record; policy?: PassRatePolicy; /** Completed weekly-run timestamps in the window, for quarantine expiry. */ weeklyRuns?: string[]; now?: number; unattributed?: string[]; errors?: string[]; manualReviews?: string[]; } const at = (record: TrialRecord) => record.recorded_at ?? ''; const runOf = (record: TrialRecord) => `${record.run_id ?? record.sha ?? 'local'}#${record.attempt}`; function seriesStats(key: string, records: TrialRecord[]): SeriesStats { const scored = records.filter(record => record.outcome !== 'skipped'); const passes = scored.filter(record => record.outcome === 'passed').length; const times = records.map(at).sort(); const first = records[0]!; return { key, identity: first.series_identity ?? 'unknown', model: first.model ?? 'unknown', cli: first.cli_version ?? 'unknown', policyVersion: first.policy_version, passes, trials: scored.length, infra: scored.filter(record => record.outcome === 'failed' && record.failure_class === 'infra').length, interval: wilsonInterval(passes, scored.length), firstSeen: times[0] ?? '', lastSeen: times[times.length - 1] ?? '', runs: [...new Set(records.map(runOf))], }; } /** Weekly runs completed after an entry's enteredAt; offline, whole weeks elapsed. */ export function quarantineRunsSince(enteredAt: string, weeklyRuns: string[] | undefined, now: number): number { const entered = Date.parse(enteredAt); if (!Number.isFinite(entered)) return Number.POSITIVE_INFINITY; if (weeklyRuns && weeklyRuns.length) return weeklyRuns.filter(time => Date.parse(time) > entered).length; return Math.floor((now - entered) / (7 * 86_400_000)); } /** * Static CASE_QUARANTINE problems, shared by the free policy test and the * weekly gate: an id that is not a blocking-tier E2E case, a missing field, * a failure class outside detector / harness / model-latency (a product * defect is fixed or named, never quarantined), a malformed or future date, * and a tier over its cap. */ export function quarantinePolicyProblems(quarantine: Record, registry: Registry = LIVE_REGISTRY, policy: PassRatePolicy = EVAL_POLICY, now = Date.now()): Alarm[] { const problems: Alarm[] = []; const invalid = (id: string, message: string) => problems.push({ kind: 'quarantine-invalid', case: id, message: `${id}: ${message}` }); const perTier = new Map(); for (const [id, entry] of Object.entries(quarantine)) { const tier = registry.tiers[id]; if (!tier || !(id in registry.kinds)) { invalid(id, 'CASE_QUARANTINE names no registered E2E case'); continue; } if (!BLOCKING_TIERS.includes(tier)) invalid(id, `tier ${tier} is not blocking; only ${BLOCKING_TIERS.join(' and ')} cases are quarantined`); for (const field of ['reason', 'failureClass', 'tracking', 'owner', 'enteredAt', 'exit'] as const) { if (typeof entry[field] !== 'string' || !entry[field].trim()) invalid(id, `missing ${field}`); } if (typeof entry.reason === 'string' && entry.reason.trim().length < 40) invalid(id, 'reason must be a written diagnosis (at least 40 characters)'); if (!QUARANTINE_FAILURE_CLASSES.includes(entry.failureClass)) { invalid(id, `failureClass ${JSON.stringify(entry.failureClass)} is not ${QUARANTINE_FAILURE_CLASSES.join(', ')}; a product defect is fixed or named as a red, never quarantined`); } const entered = Date.parse(entry.enteredAt); if (!/^\d{4}-\d{2}-\d{2}$/.test(entry.enteredAt ?? '') || !Number.isFinite(entered)) invalid(id, 'enteredAt must be YYYY-MM-DD'); else if (entered > now) invalid(id, 'enteredAt is in the future'); perTier.set(tier, (perTier.get(tier) ?? 0) + 1); } for (const [tier, count] of perTier) { const size = Object.values(registry.tiers).filter(value => value === tier).length; const cap = Math.floor(size * policy.quarantine.capFraction); if (count > cap) problems.push({ kind: 'quarantine-cap', case: tier, message: `${count} quarantined ${tier} cases exceed the ${pct(policy.quarantine.capFraction)} cap (${cap} of ${size})` }); } return problems; } export function analyzePassRates(records: TrialRecord[], options: AnalyzeOptions = {}): PassRateReport { const registry = options.registry ?? LIVE_REGISTRY; const quarantine = options.quarantine ?? CASE_QUARANTINE; const policy = options.policy ?? EVAL_POLICY; const now = options.now ?? Date.now(); const byCase = new Map(); for (const record of records) { const list = byCase.get(record.case) ?? []; list.push(record); byCase.set(record.case, list); } for (const id of options.manualReviews ?? []) if (!byCase.has(id)) byCase.set(id, []); const cases: CasePassRate[] = []; for (const [id, list] of [...byCase].sort(([a], [b]) => a.localeCompare(b))) { list.sort((a, b) => at(a).localeCompare(at(b)) || runOf(a).localeCompare(runOf(b)) || a.trial - b.trial); const groups = new Map(); for (const record of list) { const key = record.policy_version === 0 ? 'pre-policy' : [record.series_identity ?? 'unknown', record.model ?? 'unknown', record.cli_version ?? 'unknown', `v${record.policy_version}`].join('|'); const group = groups.get(key) ?? []; group.push(record); groups.set(key, group); } const series = [...groups].map(([key, group]) => seriesStats(key, group)) .sort((a, b) => a.lastSeen.localeCompare(b.lastSeen)); const post = series.filter(entry => entry.policyVersion !== 0); const current = post[post.length - 1] ?? null; const previous = post[post.length - 2] ?? null; const scored = current ? groups.get(current.key)!.filter(record => record.outcome !== 'skipped') : []; const latestRun = scored.length ? runOf(scored[scored.length - 1]!) : null; const latest = scored.filter(record => runOf(record) === latestRun); const prior = scored.filter(record => runOf(record) !== latestRun); const priorPasses = prior.filter(record => record.outcome === 'passed').length; const entryRate = policy.quarantine.entry.rate; let label: PassRateLabel; if (latest.length > 0 && latest.every(record => record.outcome === 'failed') && prior.length > 0 && wilsonInterval(priorPasses, prior.length).lo >= entryRate) label = 'BROKEN'; else if (!current || current.trials < policy.quarantine.entry.minTrials) label = 'INCONCLUSIVE'; else if (current.interval.hi < entryRate) label = 'FAILING'; else if (current.passes < current.trials && current.interval.lo < entryRate) label = 'FLAKY'; else label = 'PASSING'; cases.push({ case: id, kind: registry.kinds[id] ?? list[0]!.kind, tier: caseTier(id, registry), quarantined: id in quarantine, label, current, previous, manualReviews: (options.manualReviews ?? []).filter(name => name === id).length, prePolicy: series.find(entry => entry.policyVersion === 0) ?? null, series, latestRun: latestRun ? { runId: latestRun, passes: latest.filter(record => record.outcome === 'passed').length, trials: latest.length } : null, }); } const alarms: Alarm[] = []; const rate = (stats: SeriesStats) => stats.passes / stats.trials; for (const entry of cases) { const current = entry.current; if (!current) continue; const below = current.trials >= policy.quarantine.entry.minTrials && rate(current) < policy.quarantine.entry.rate; if (below && !entry.quarantined && BLOCKING_TIERS.includes(entry.tier)) alarms.push({ kind: 'drift', case: entry.case, message: `${entry.case} passes ${current.passes}/${current.trials} (below ${pct(policy.quarantine.entry.rate)} over >= ${policy.quarantine.entry.minTrials} trials): fix it, or propose a CASE_QUARANTINE entry with a written diagnosis (product defects are never quarantined)` }); if (below && entry.kind === 'rule') alarms.push({ kind: 'rule-as-behavior', case: entry.case, message: `${entry.case}: rule case behaving like behavior (${current.passes}/${current.trials}): fix or reclassify` }); if (entry.quarantined && current.trials >= policy.quarantine.exit.minTrials && rate(current) >= policy.quarantine.exit.rate) { alarms.push({ kind: 'quarantine-exit', case: entry.case, message: `${entry.case} passes ${current.passes}/${current.trials} (>= ${pct(policy.quarantine.exit.rate)}): remove its CASE_QUARANTINE entry` }); } } const tested = cases.filter(entry => BLOCKING_TIERS.includes(entry.tier) && entry.current && entry.previous && entry.current.trials >= policy.drift.fisherMinPerSide && entry.previous.trials >= policy.drift.fisherMinPerSide); const pValues = tested.map(entry => fisherOneSidedLower(entry.current!.passes, entry.current!.trials, entry.previous!.passes, entry.previous!.trials)); for (const index of holmRejections(pValues, policy.drift.fisherAlpha)) { const entry = tested[index]!; alarms.push({ kind: 'regression', case: entry.case, message: `${entry.case}: current identity ${entry.current!.passes}/${entry.current!.trials} is significantly below the previous ${entry.previous!.passes}/${entry.previous!.trials} (one-sided Fisher p=${pValues[index]!.toFixed(4)}, Holm over ${tested.length} cases)` }); } for (const [id, entry] of Object.entries(quarantine)) { const runs = quarantineRunsSince(entry.enteredAt, options.weeklyRuns, now); if (runs >= policy.quarantine.expiryWeeklyRuns) alarms.push({ kind: 'quarantine-expired', case: id, message: `${id}: entered ${runs} weekly runs ago (limit ${policy.quarantine.expiryWeeklyRuns}): fix it, name it as a red, or re-diagnose with fresh evidence` }); } alarms.push(...quarantinePolicyProblems(quarantine, registry, policy, now)); const post = records.filter(record => record.policy_version !== 0).length; return { policyVersion: policy.version, cases, alarms, postPolicyTrials: post, prePolicyTrials: records.length - post, unattributed: options.unattributed ?? [], errors: options.errors ?? [] }; } function pct(value: number): string { return `${Math.round(value * 1000) / 10}%`; } function formatStats(stats: SeriesStats | null): string { if (!stats) return '-'; return `${stats.passes}/${stats.trials} [${pct(stats.interval.lo)}–${pct(stats.interval.hi)}]${stats.infra ? ` (${stats.infra} infra)` : ''}`; } export function formatPassRates(report: PassRateReport, options: { caseFilter?: string } = {}): string { const lines: string[] = []; lines.push(`pass-rates: policy v${report.policyVersion}, ${report.postPolicyTrials} post-policy trial(s), ${report.prePolicyTrials} pre-policy (display only)`); if (report.postPolicyTrials === 0) lines.push(' no post-policy trials yet: every series starts INCONCLUSIVE'); const cases = report.cases.filter(entry => !options.caseFilter || entry.case === options.caseFilter); lines.push(' label kind tier current series pre-policy manual case'); for (const entry of cases) { const group = entry.current ? ` ${entry.current.model} / ${entry.current.cli}` : ''; const reset = entry.previous ? ' (baseline reset)' : ''; lines.push(` ${entry.label.padEnd(12)} ${entry.kind.padEnd(8)} ${entry.tier.padEnd(8)} ${formatStats(entry.current).padEnd(29)} ` + `${formatStats(entry.prePolicy).padEnd(18)} ${String(entry.manualReviews).padStart(6)} ${entry.case}${entry.quarantined ? ' [quarantined]' : ''}${group}${reset}`); } if (report.unattributed.length) lines.push(` unattributed records (${report.unattributed.length}, never guessed): ${report.unattributed.slice(0, 20).join(', ')}`); if (report.errors.length) lines.push(` rejected ${report.errors.length} invalid trial line(s): ${report.errors.slice(0, 5).join('; ')}`); if (report.alarms.length) { lines.push(`ACTION REQUIRED (${report.alarms.length}):`); for (const alarm of report.alarms) lines.push(` [${alarm.kind}] ${alarm.message}`); } return lines.join('\n'); } // --- GitHub history --- export interface WeeklyRun { id: number; attempt: number; sha: string; branch: string; createdAt: string } export interface RunArtifact { id: number; name: string; size: number } /** The GitHub calls pass-rates makes; injectable so the free tests never touch the network. */ export interface HistoryFetcher { listRuns(repo: string, workflow: string, branch: string, limit: number): WeeklyRun[]; listArtifacts(repo: string, runId: number): RunArtifact[]; downloadZip(repo: string, artifactId: number, destination: string): void; } function gh(args: string[]): Buffer { const result = spawnSync('gh', args, { timeout: 300_000, maxBuffer: 256 * 1024 * 1024 }); if (result.status !== 0) throw new Error(`gh ${args.slice(0, 2).join(' ')} failed: ${String(result.stderr || result.error || '').trim()}`); return result.stdout; } function jsonLines(buffer: Buffer): T[] { return buffer.toString('utf8').split('\n').filter(Boolean).map(line => JSON.parse(line) as T); } export const GH_HISTORY: HistoryFetcher = { listRuns: (repo, workflow, branch, limit): WeeklyRun[] => jsonLines(gh(['api', `repos/${repo}/actions/workflows/${workflow}/runs?branch=${encodeURIComponent(branch)}&status=completed&per_page=${limit}`, '--jq', '.workflow_runs[] | {id, attempt: .run_attempt, sha: .head_sha, branch: .head_branch, createdAt: .created_at}'])), listArtifacts: (repo, runId): RunArtifact[] => jsonLines(gh(['api', `repos/${repo}/actions/runs/${runId}/artifacts?per_page=100`, '--paginate', '--jq', '.artifacts[] | select(.expired | not) | {id, name, size: .size_in_bytes}'])), downloadZip: (repo, artifactId, destination) => fs.writeFileSync(destination, gh(['api', `repos/${repo}/actions/artifacts/${artifactId}/zip`])), }; /** The last `limit` completed runs of `workflow` on each branch, newest first, deduplicated. */ export function listWeeklyRuns(opts: { repo: string; workflow: string; branches: string[]; limit: number; fetcher?: HistoryFetcher }): WeeklyRun[] { const fetcher = opts.fetcher ?? GH_HISTORY; const runs = new Map(); for (const branch of opts.branches) for (const run of fetcher.listRuns(opts.repo, opts.workflow, branch, opts.limit)) runs.set(run.id, run); return [...runs.values()].sort((a, b) => b.createdAt.localeCompare(a.createdAt)); } /** * Download the artifacts of one run whose names match into a per-run cache * directory (reused on later calls) and return the extracted directories. * Oversized or oddly named artifacts are skipped: downloads are data only. */ export function downloadRunArtifacts(opts: { repo: string; run: WeeklyRun; match: (name: string) => boolean; cacheDir: string; fetcher?: HistoryFetcher; maxBytes?: number }): string[] { const fetcher = opts.fetcher ?? GH_HISTORY; const dirs: string[] = []; for (const artifact of fetcher.listArtifacts(opts.repo, opts.run.id)) { if (!opts.match(artifact.name) || !/^[A-Za-z0-9._-]+$/.test(artifact.name)) continue; if (artifact.size > (opts.maxBytes ?? TRIAL_OUTCOMES_MAX_BYTES)) continue; const dir = path.join(opts.cacheDir, `${opts.run.id}`, artifact.name); if (!fs.existsSync(path.join(dir, '.complete'))) { fs.rmSync(dir, { recursive: true, force: true }); fs.mkdirSync(dir, { recursive: true }); const zip = path.join(dir, 'artifact.zip'); fetcher.downloadZip(opts.repo, artifact.id, zip); const unzip = spawnSync('unzip', ['-o', '-q', zip, '-d', dir], { timeout: 120_000 }); if (unzip.status !== 0) throw new Error(`unzip failed for ${artifact.name}: ${String(unzip.stderr || unzip.error || '')}`); fs.rmSync(zip, { force: true }); fs.writeFileSync(path.join(dir, '.complete'), ''); } dirs.push(dir); } return dirs; } function gitOutput(args: string[]): string | null { const result = spawnSync('git', args, { encoding: 'utf8', timeout: 5_000 }); return result.status === 0 ? result.stdout.trim() : null; } function repoSlug(): string { const url = gitOutput(['remote', 'get-url', 'origin']) ?? ''; return url.match(/[:/]([^/:]+\/[^/]+?)(?:\.git)?$/)?.[1] ?? 'garrytan/gstack'; } if (import.meta.main) { const argv = process.argv.slice(2); const flag = (name: string) => { const index = argv.indexOf(name); return index === -1 ? undefined : argv[index + 1]; }; const dirs = argv.flatMap((arg, index) => arg === '--dir' && argv[index + 1] ? [argv[index + 1]!] : []); const asJson = argv.includes('--json'); const gate = argv.includes('--gate'); const backfill = argv.includes('--backfill'); const caseFilter = flag('--case'); const runsLimit = Number(flag('--runs')) || 10; const sinceDays = Number(flag('--since-days')) || 60; const repo = flag('--repo') ?? repoSlug(); const workflow = flag('--workflow') ?? 'evals-periodic.yml'; const branch = flag('--branch') ?? gitOutput(['rev-parse', '--abbrev-ref', 'HEAD']) ?? 'main'; const records: TrialRecord[] = []; const unattributed = new Set(); const errors: string[] = []; let historyError: string | null = null; let weeklyRuns: string[] | undefined; const manualReviews: string[] = []; const importDir = (dir: string, run: { run_id: string; sha?: string; timestamp?: string } | undefined, legacyDays: number) => { const trials = readTrialOutcomeDir(dir); records.push(...trials.records); errors.push(...trials.errors); const legacy = backfillEvalFiles(collectEvalFiles(dir, legacyDays), run); records.push(...legacy.records); manualReviews.push(...legacy.manualReviews); legacy.unattributed.forEach(name => unattributed.add(name)); }; if (dirs.length) { for (const dir of dirs) importDir(dir, undefined, sinceDays); } else { try { const runs = listWeeklyRuns({ repo, workflow, branches: [...new Set([branch, 'main'])], limit: runsLimit }); weeklyRuns = runs.map(run => run.createdAt); const cacheDir = path.join(os.homedir(), '.gstack', 'eval-pass-rates-cache', repo.replace('/', '-')); const match = backfill ? (name: string) => name.startsWith('trial-outcomes') || /^(paid-slice-\d+|gate-census-\d+)(-a\d+)?$/.test(name) : (name: string) => name.startsWith('trial-outcomes'); for (const run of runs) { const dirsForRun = downloadRunArtifacts({ repo, run, match, cacheDir, maxBytes: backfill ? 64 * 1024 * 1024 : undefined }); for (const dir of dirsForRun) importDir(dir, { run_id: `${run.id}`, sha: run.sha, timestamp: run.createdAt }, 3650); } } catch (error) { historyError = error instanceof Error ? error.message : String(error); } } const report = analyzePassRates(records, { weeklyRuns, unattributed: [...unattributed].sort(), errors, manualReviews }); const ledger = readFreeLedger(); if (asJson) { console.log(JSON.stringify({ repo, workflow, branch, dirs, historyError, ...report, freeLedger: ledger }, null, 2)); } else { if (historyError) console.log(`pass-rates: history unavailable (${historyError}); every label below is INCONCLUSIVE`); console.log(formatPassRates(report, { caseFilter })); if (ledger.length > 0) { const byFile = new Map(); for (const e of ledger) byFile.set(e.file, (byFile.get(e.file) ?? 0) + 1); console.log(`free-suite flaky-passes (${flakeLedgerPath()}):`); for (const [file, n] of [...byFile.entries()].sort((a, b) => b[1] - a[1])) { console.log(` ${String(n).padStart(3)}x ${file}`); } } } if (gate && (historyError || report.alarms.length)) process.exit(1); }