Files
gstack/scripts/eval-flake-rank.ts
T

706 lines
35 KiB
TypeScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env bun
/**
* eval-pass-rates (alias: eval-flake-rank) — per-case trial pass rates.
*
* Reads trial records (one JSONL line per trial: case, kind, trial, outcome,
* exit_reason, duration, cost, model, CLI version, series identity, run id,
* sha, policy_version) from the last N completed `evals-periodic.yml` runs on
* the current branch and `main` (downloading only each run's small
* `trial-outcomes` artifact through `gh`), plus any local eval dirs, and
* prints per-case per-trial pass rates with 95% Wilson intervals.
*
* A series is one case under one input identity: the case's own touchfiles
* minus GLOBAL_TOUCHFILES (`caseSeriesIdentities`), grouped by model and CLI
* version, per policy_version. A new identity starts a new series; earlier
* series stay visible. Only post-policy trials of the current series feed the
* labels and alarms. Legacy eval-store records (`--backfill`, `--dir`) are
* imported as pre-policy trials (first attempt only; a missing attempt means
* 1) and are display-only.
*
* Labels: INCONCLUSIVE (below the entry rule's minimum trials), BROKEN (latest run 0/n
* after a prior interval at or above the entry rate), FLAKY (failures and an
* interval straddling the entry rate), FAILING (interval below the entry
* rate), PASSING (otherwise).
*
* The weekly gate (`--gate`) exits non-zero with ACTION REQUIRED when a
* non-quarantined case meets the quarantine entry rule, a rule case behaves
* like a behavior case, a blocking case's current-identity rate is
* significantly below its previous identity (one-sided Fisher exact,
* Holm-controlled across cases), or a CASE_QUARANTINE entry has met its exit
* rule, expired, or pushed its tier over the cap. History that cannot be
* fetched fails the gate closed.
*
* Usage:
* bun run eval:pass-rates # last 10 weekly runs, this branch + main
* bun run eval:pass-rates --case <id> --runs 20
* bun run eval:pass-rates --dir <path> # local eval dirs / downloaded artifacts (repeatable)
* bun run eval:pass-rates --backfill # also import legacy slice artifacts, labeled pre-policy
* bun run eval:pass-rates --json | --gate
*/
import * as fs from 'node:fs';
import * as os from 'node:os';
import * as path from 'node:path';
import { spawnSync } from 'node:child_process';
import { createHash } from 'node:crypto';
import { isPartialEval, isFinalizedEvalResultFile, evalEntryOutcome, failureClassOf, parseTrialOutcomes, sanitizeTrialError,
TRIAL_OUTCOME_SCHEMA, type EvalCaseKind, type EvalResult, type TrialOutcomeRecord } from '../test/helpers/eval-store';
import { flakeLedgerPath, type FlakeLedgerEntry } from './test-free-shards';
import { E2E_KINDS, E2E_TIERS, E2E_TOUCHFILES, GLOBAL_TOUCHFILES, LLM_JUDGE_TOUCHFILES } from '../test/helpers/touchfiles-data';
import { CASE_QUARANTINE, EVAL_POLICY } from '../test/helpers/periodic-exclude-data';
import { matchGlob } from '../test/helpers/test-selection';
import { CASE_TEST_NAMES } from './test-paid-shards';
interface TestSeries {
name: string;
runs: number;
passes: number;
fails: number;
manualAccepted: number;
retriedPasses: number;
totalAttempts: number;
totalCostUsd: number;
totalDurationMs: number;
lastSeen: string;
}
export function aggregate(evalFiles: string[]): Map<string, TestSeries> {
const series = new Map<string, TestSeries>();
for (const file of evalFiles) {
let run: EvalResult;
try {
run = JSON.parse(fs.readFileSync(file, 'utf-8'));
} catch { continue; }
if (isPartialEval(run, file)) continue; // in-progress accumulators are not runs
if (!Array.isArray(run.tests)) continue;
// Group this run's entries by name so N attempts = 1 run of that test.
const byName = new Map<string, typeof run.tests>();
for (const t of run.tests) {
const list = byName.get(t.name) ?? [];
list.push(t);
byName.set(t.name, list);
}
for (const [name, entries] of byName) {
const s = series.get(name) ?? {
name, runs: 0, passes: 0, fails: 0, manualAccepted: 0, retriedPasses: 0,
totalAttempts: 0, totalCostUsd: 0, totalDurationMs: 0, lastSeen: '',
};
const final = entries[entries.length - 1];
s.totalAttempts += entries.length;
const outcome = evalEntryOutcome(final);
if (outcome === 'manual-review') s.manualAccepted += 1;
else {
s.runs += 1;
if (outcome === 'passed') s.passes += 1; else s.fails += 1;
if (outcome === 'passed' && entries.length > 1) s.retriedPasses += 1;
}
for (const e of entries) {
s.totalCostUsd += e.cost_usd || 0;
s.totalDurationMs += e.duration_ms || 0;
}
if (run.timestamp > s.lastSeen) s.lastSeen = run.timestamp;
series.set(name, s);
}
}
return series;
}
export function collectEvalFiles(dir: string, sinceDays = 60): string[] {
if (!fs.existsSync(dir)) return [];
const cutoff = Date.now() - sinceDays * 86_400_000;
const out: string[] = [];
for (const name of fs.readdirSync(dir, { recursive: true }) as string[]) {
if (!isFinalizedEvalResultFile(name)) continue;
const full = path.join(dir, name);
try {
// Recency bound (review finding): E2E results embed full transcripts
// (MBs each) and the scan is otherwise unbounded over all-time history.
if (fs.statSync(full).mtimeMs < cutoff) continue;
} catch { continue; }
out.push(full);
}
return out;
}
function readFreeLedger(): FlakeLedgerEntry[] {
// Per-LINE parse: one malformed JSONL line (torn write, manual edit) must
// drop that line, never vanish the whole series (codex adversarial finding).
let raw: string;
try {
raw = fs.readFileSync(flakeLedgerPath(), 'utf-8');
} catch { return []; }
const out: FlakeLedgerEntry[] = [];
for (const line of raw.split('\n')) {
if (!line.trim()) continue;
try { out.push(JSON.parse(line)); } catch { /* torn line — skip */ }
}
return out;
}
// --- Trial records ---
/**
* A trial record as pass-rates reads it: eval-store's trial-outcomes schema
* plus the series identity the report job stamps (caseSeriesIdentities).
* policy_version 0 marks a pre-policy (backfilled) record.
*/
export type TrialRecord = TrialOutcomeRecord & { series_identity?: string };
/** Per-file cap for downloaded artifacts: pass-rates parses data only, never executes it. */
export const TRIAL_OUTCOMES_MAX_BYTES = 8 * 1024 * 1024;
/** Every `trial-outcomes*.jsonl` file under a directory, size-capped, schema-validated by eval-store. */
export function readTrialOutcomeDir(dir: string): { records: TrialRecord[]; errors: string[] } {
const records: TrialRecord[] = [];
const errors: string[] = [];
if (!fs.existsSync(dir)) return { records, errors };
for (const name of fs.readdirSync(dir, { recursive: true }) as string[]) {
if (!/(^|\/)trial-outcomes[^/]*\.jsonl$/.test(name)) continue;
const full = path.join(dir, name);
const parsed = parseTrialOutcomes(fs.readFileSync(full, 'utf8'), { maxBytes: TRIAL_OUTCOMES_MAX_BYTES });
records.push(...parsed.records.map(record => ({
...record, series_identity: typeof (record as TrialRecord).series_identity === 'string'
? (record as TrialRecord).series_identity!.slice(0, 64) : undefined })));
errors.push(...parsed.errors.map(error => `${full}: ${error}`));
}
return { records, errors };
}
// --- Registry attribution and series identity ---
export interface Registry {
kinds: Record<string, EvalCaseKind>;
tiers: Record<string, string>;
touchfiles: Record<string, string[]>;
judgeTouchfiles: Record<string, string[]>;
globals: readonly string[];
testNames: Record<string, string>;
}
export const LIVE_REGISTRY: Registry = {
kinds: E2E_KINDS, tiers: E2E_TIERS, touchfiles: E2E_TOUCHFILES, judgeTouchfiles: LLM_JUDGE_TOUCHFILES,
globals: GLOBAL_TOUCHFILES, testNames: CASE_TEST_NAMES,
};
/** A case's tier: its E2E_TIERS value, or 'judge' for an LLM-judge entry. */
export function caseTier(id: string, registry: Registry = LIVE_REGISTRY): string {
return registry.tiers[id] ?? (id in registry.judgeTouchfiles ? 'judge' : 'unknown');
}
/**
* Attribute a legacy eval-store record to a registry id: the case-shard slug
* suffix (`<file>--<id>`), the recorded name or its exact slug (`/qa b6-static`
* is `qa-b6-static`), a CASE_TEST_NAMES label, or the only id its shard file
* registers. Anything else is unattributed (null).
*/
export function attributeLegacyRecord(name: string, shard: string | undefined, registry: Registry = LIVE_REGISTRY): string | null {
const known = (id: string) => id in registry.kinds;
const [slugFile, slugCase] = (shard ?? '').split('--');
if (slugCase && known(slugCase)) return slugCase;
if (known(name)) return name;
const slug = name.toLowerCase().replace(/[^a-z0-9]+/g, '-').replace(/^-+|-+$/g, '');
if (known(slug)) return slug;
const labeled = Object.entries(registry.testNames).find(([, label]) => label === name)?.[0];
if (labeled && known(labeled)) return labeled;
if (slugFile) {
const file = `test/${slugFile}.test.ts`;
const owners = Object.keys(registry.touchfiles).filter(id => registry.touchfiles[id]!.includes(file));
if (owners.length === 1 && known(owners[0]!)) return owners[0]!;
}
return null;
}
/**
* Series identity per case: a hash of the git blob ids of the files matching
* the case's own touchfiles, excluding GLOBAL_TOUCHFILES (harness edits are
* markers, not new series). The report job stamps this on every trial record.
*/
export function caseSeriesIdentities(ids: string[], root: string, registry: Registry = LIVE_REGISTRY): Record<string, string> {
const listed = spawnSync('git', ['ls-files', '-s'], { cwd: root, encoding: 'utf8', timeout: 20_000, maxBuffer: 64 * 1024 * 1024 });
if (listed.status !== 0) throw new Error(`git ls-files failed: ${listed.stderr}`);
const blobs = listed.stdout.split('\n').filter(Boolean).map(line => {
const [meta, file] = line.split('\t');
return { file: file!, blob: meta!.split(' ')[1]! };
}).filter(entry => !registry.globals.some(pattern => matchGlob(entry.file, pattern)));
return Object.fromEntries(ids.map(id => {
const patterns = registry.touchfiles[id] ?? registry.judgeTouchfiles[id] ?? [];
const lines = blobs.filter(entry => patterns.some(pattern => matchGlob(entry.file, pattern)))
.map(entry => `${entry.file} ${entry.blob}`).sort();
return [id, createHash('sha256').update(`${id}\n${lines.join('\n')}`).digest('hex').slice(0, 16)];
}));
}
/**
* Import legacy eval-store result files as pre-policy trials (policy_version
* 0, source 'backfill'): first attempt only (a missing attempt means 1),
* attributed by registry id, never guessed. A manual-review acceptance carries
* no automated verdict: it is counted and shown, never scored. Without a CI
* run, each local result file is its own run.
*/
export function backfillEvalFiles(files: string[], run?: { run_id: string; sha?: string; timestamp?: string },
registry: Registry = LIVE_REGISTRY): { records: TrialRecord[]; unattributed: string[]; manualReviews: string[] } {
const records: TrialRecord[] = [];
const unattributed = new Set<string>();
const manualReviews: string[] = [];
for (const file of files) {
let result: EvalResult & { shard?: string; claude_cli_version?: string };
try { result = JSON.parse(fs.readFileSync(file, 'utf8')); } catch { continue; }
if (isPartialEval(result, file) || !Array.isArray(result.tests)) continue;
const seen = new Set<string>();
for (const entry of result.tests) {
if ((entry.attempt ?? 1) !== 1 || seen.has(entry.name)) continue;
seen.add(entry.name);
const id = attributeLegacyRecord(entry.name, result.shard, registry);
if (!id) { unattributed.add(entry.name); continue; }
const outcome = evalEntryOutcome(entry);
if (outcome === 'manual-review') { manualReviews.push(id); continue; }
records.push({
schema: TRIAL_OUTCOME_SCHEMA, case: id,
file: result.shard ? `test/${result.shard.split('--')[0]}.test.ts` : 'unknown',
tier: caseTier(id, registry), kind: registry.kinds[id]!, trial: 1, panel: { n: 1, k: 1 }, attempt: 1,
outcome, ...(outcome === 'failed' ? { failure_class: failureClassOf(entry) } : {}),
exit_reason: entry.exit_reason, error: sanitizeTrialError(entry.error),
duration_ms: Math.max(0, entry.duration_ms || 0), cost_usd: Math.max(0, entry.cost_usd || 0),
model: entry.model, cli_version: result.claude_cli_version, policy_version: 0, quarantined: false,
execution: entry.execution === 'reused' ? 'reused' : 'executed', source: 'backfill',
run_id: run?.run_id ?? `local:${file}`, sha: run?.sha ?? result.git_sha, recorded_at: run?.timestamp ?? result.timestamp,
});
}
}
return { records, unattributed: [...unattributed].sort(), manualReviews };
}
// --- Statistics ---
/** 95% Wilson score interval for k successes in n trials. */
export function wilsonInterval(k: number, n: number, z = 1.96): { lo: number; hi: number } {
if (n <= 0) return { lo: 0, hi: 1 };
const p = k / n, z2 = z * z, denom = 1 + z2 / n;
const center = (p + z2 / (2 * n)) / denom;
const half = (z * Math.sqrt(p * (1 - p) / n + z2 / (4 * n * n))) / denom;
return { lo: Math.max(0, center - half), hi: k === n ? 1 : Math.min(1, center + half) };
}
function logChoose(n: number, k: number): number {
let sum = 0;
for (let i = 1; i <= k; i++) sum += Math.log(n - k + i) - Math.log(i);
return sum;
}
/**
* One-sided Fisher exact p-value that the CURRENT pass rate is below the
* PREVIOUS one: P(X <= curPass) under the hypergeometric null with the
* observed margins.
*/
export function fisherOneSidedLower(curPass: number, curN: number, prevPass: number, prevN: number): number {
const passes = curPass + prevPass, total = curN + prevN;
const denom = logChoose(total, passes);
let p = 0;
for (let x = Math.max(0, passes - prevN); x <= curPass; x++) p += Math.exp(logChoose(curN, x) + logChoose(prevN, passes - x) - denom);
return Math.min(1, p);
}
/** Holm step-down: the indices whose p-values are rejected at family-wise alpha. */
export function holmRejections(pValues: number[], alpha: number): Set<number> {
const order = pValues.map((p, index) => ({ p, index })).sort((a, b) => a.p - b.p);
const rejected = new Set<number>();
for (let rank = 0; rank < order.length; rank++) {
if (order[rank]!.p > alpha / (order.length - rank)) break;
rejected.add(order[rank]!.index);
}
return rejected;
}
// --- Analysis ---
export type PassRateLabel = 'INCONCLUSIVE' | 'BROKEN' | 'FLAKY' | 'FAILING' | 'PASSING';
export type AlarmKind = 'drift' | 'rule-as-behavior' | 'regression' | 'quarantine-exit' | 'quarantine-expired'
| 'quarantine-cap' | 'quarantine-invalid';
/** The EVAL_POLICY fields pass-rates reads (structural, so tests can vary them). */
export interface PassRatePolicy {
version: number;
quarantine: { entry: { rate: number; minTrials: number }; exit: { rate: number; minTrials: number }; capFraction: number; expiryWeeklyRuns: number };
drift: { fisherAlpha: number; fisherMinPerSide: number };
}
export type QuarantineEntry = (typeof CASE_QUARANTINE)[string];
/** Tiers whose cases block a lane; quarantine applies only to them. */
export const BLOCKING_TIERS: readonly string[] = ['gate', 'periodic'];
const QUARANTINE_FAILURE_CLASSES: readonly string[] = ['detector', 'harness', 'model-latency'];
export interface SeriesStats {
key: string;
identity: string;
model: string;
cli: string;
policyVersion: number;
passes: number;
/** Scored trials: passed + failed (skipped trials carry no verdict). */
trials: number;
infra: number;
interval: { lo: number; hi: number };
firstSeen: string;
lastSeen: string;
runs: string[];
}
export interface CasePassRate {
case: string;
kind: EvalCaseKind;
tier: string;
quarantined: boolean;
label: PassRateLabel;
/** Manual-review acceptances: visible, never scored. */
manualReviews: number;
current: SeriesStats | null;
previous: SeriesStats | null;
prePolicy: SeriesStats | null;
series: SeriesStats[];
latestRun: { runId: string; passes: number; trials: number } | null;
}
export interface Alarm { kind: AlarmKind; case: string; message: string }
export interface PassRateReport {
policyVersion: number;
cases: CasePassRate[];
alarms: Alarm[];
postPolicyTrials: number;
prePolicyTrials: number;
unattributed: string[];
errors: string[];
}
export interface AnalyzeOptions {
registry?: Registry;
quarantine?: Record<string, QuarantineEntry>;
policy?: PassRatePolicy;
/** Completed weekly-run timestamps in the window, for quarantine expiry. */
weeklyRuns?: string[];
now?: number;
unattributed?: string[];
errors?: string[];
manualReviews?: string[];
}
const at = (record: TrialRecord) => record.recorded_at ?? '';
const runOf = (record: TrialRecord) => `${record.run_id ?? record.sha ?? 'local'}#${record.attempt}`;
function seriesStats(key: string, records: TrialRecord[]): SeriesStats {
const scored = records.filter(record => record.outcome !== 'skipped');
const passes = scored.filter(record => record.outcome === 'passed').length;
const times = records.map(at).sort();
const first = records[0]!;
return {
key, identity: first.series_identity ?? 'unknown', model: first.model ?? 'unknown', cli: first.cli_version ?? 'unknown',
policyVersion: first.policy_version, passes, trials: scored.length,
infra: scored.filter(record => record.outcome === 'failed' && record.failure_class === 'infra').length,
interval: wilsonInterval(passes, scored.length), firstSeen: times[0] ?? '', lastSeen: times[times.length - 1] ?? '',
runs: [...new Set(records.map(runOf))],
};
}
/** Weekly runs completed after an entry's enteredAt; offline, whole weeks elapsed. */
export function quarantineRunsSince(enteredAt: string, weeklyRuns: string[] | undefined, now: number): number {
const entered = Date.parse(enteredAt);
if (!Number.isFinite(entered)) return Number.POSITIVE_INFINITY;
if (weeklyRuns && weeklyRuns.length) return weeklyRuns.filter(time => Date.parse(time) > entered).length;
return Math.floor((now - entered) / (7 * 86_400_000));
}
/**
* Static CASE_QUARANTINE problems, shared by the free policy test and the
* weekly gate: an id that is not a blocking-tier E2E case, a missing field,
* a failure class outside detector / harness / model-latency (a product
* defect is fixed or named, never quarantined), a malformed or future date,
* and a tier over its cap.
*/
export function quarantinePolicyProblems(quarantine: Record<string, QuarantineEntry>,
registry: Registry = LIVE_REGISTRY, policy: PassRatePolicy = EVAL_POLICY, now = Date.now()): Alarm[] {
const problems: Alarm[] = [];
const invalid = (id: string, message: string) => problems.push({ kind: 'quarantine-invalid', case: id, message: `${id}: ${message}` });
const perTier = new Map<string, number>();
for (const [id, entry] of Object.entries(quarantine)) {
const tier = registry.tiers[id];
if (!tier || !(id in registry.kinds)) { invalid(id, 'CASE_QUARANTINE names no registered E2E case'); continue; }
if (!BLOCKING_TIERS.includes(tier)) invalid(id, `tier ${tier} is not blocking; only ${BLOCKING_TIERS.join(' and ')} cases are quarantined`);
for (const field of ['reason', 'failureClass', 'tracking', 'owner', 'enteredAt', 'exit'] as const) {
if (typeof entry[field] !== 'string' || !entry[field].trim()) invalid(id, `missing ${field}`);
}
if (typeof entry.reason === 'string' && entry.reason.trim().length < 40) invalid(id, 'reason must be a written diagnosis (at least 40 characters)');
if (!QUARANTINE_FAILURE_CLASSES.includes(entry.failureClass)) {
invalid(id, `failureClass ${JSON.stringify(entry.failureClass)} is not ${QUARANTINE_FAILURE_CLASSES.join(', ')}; a product defect is fixed or named as a red, never quarantined`);
}
const entered = Date.parse(entry.enteredAt);
if (!/^\d{4}-\d{2}-\d{2}$/.test(entry.enteredAt ?? '') || !Number.isFinite(entered)) invalid(id, 'enteredAt must be YYYY-MM-DD');
else if (entered > now) invalid(id, 'enteredAt is in the future');
perTier.set(tier, (perTier.get(tier) ?? 0) + 1);
}
for (const [tier, count] of perTier) {
const size = Object.values(registry.tiers).filter(value => value === tier).length;
const cap = Math.floor(size * policy.quarantine.capFraction);
if (count > cap) problems.push({ kind: 'quarantine-cap', case: tier,
message: `${count} quarantined ${tier} cases exceed the ${pct(policy.quarantine.capFraction)} cap (${cap} of ${size})` });
}
return problems;
}
export function analyzePassRates(records: TrialRecord[], options: AnalyzeOptions = {}): PassRateReport {
const registry = options.registry ?? LIVE_REGISTRY;
const quarantine = options.quarantine ?? CASE_QUARANTINE;
const policy = options.policy ?? EVAL_POLICY;
const now = options.now ?? Date.now();
const byCase = new Map<string, TrialRecord[]>();
for (const record of records) {
const list = byCase.get(record.case) ?? [];
list.push(record);
byCase.set(record.case, list);
}
for (const id of options.manualReviews ?? []) if (!byCase.has(id)) byCase.set(id, []);
const cases: CasePassRate[] = [];
for (const [id, list] of [...byCase].sort(([a], [b]) => a.localeCompare(b))) {
list.sort((a, b) => at(a).localeCompare(at(b)) || runOf(a).localeCompare(runOf(b)) || a.trial - b.trial);
const groups = new Map<string, TrialRecord[]>();
for (const record of list) {
const key = record.policy_version === 0 ? 'pre-policy'
: [record.series_identity ?? 'unknown', record.model ?? 'unknown', record.cli_version ?? 'unknown', `v${record.policy_version}`].join('|');
const group = groups.get(key) ?? [];
group.push(record);
groups.set(key, group);
}
const series = [...groups].map(([key, group]) => seriesStats(key, group))
.sort((a, b) => a.lastSeen.localeCompare(b.lastSeen));
const post = series.filter(entry => entry.policyVersion !== 0);
const current = post[post.length - 1] ?? null;
const previous = post[post.length - 2] ?? null;
const scored = current ? groups.get(current.key)!.filter(record => record.outcome !== 'skipped') : [];
const latestRun = scored.length ? runOf(scored[scored.length - 1]!) : null;
const latest = scored.filter(record => runOf(record) === latestRun);
const prior = scored.filter(record => runOf(record) !== latestRun);
const priorPasses = prior.filter(record => record.outcome === 'passed').length;
const entryRate = policy.quarantine.entry.rate;
let label: PassRateLabel;
if (latest.length > 0 && latest.every(record => record.outcome === 'failed')
&& prior.length > 0 && wilsonInterval(priorPasses, prior.length).lo >= entryRate) label = 'BROKEN';
else if (!current || current.trials < policy.quarantine.entry.minTrials) label = 'INCONCLUSIVE';
else if (current.interval.hi < entryRate) label = 'FAILING';
else if (current.passes < current.trials && current.interval.lo < entryRate) label = 'FLAKY';
else label = 'PASSING';
cases.push({
case: id, kind: registry.kinds[id] ?? list[0]!.kind, tier: caseTier(id, registry),
quarantined: id in quarantine, label, current, previous,
manualReviews: (options.manualReviews ?? []).filter(name => name === id).length,
prePolicy: series.find(entry => entry.policyVersion === 0) ?? null, series,
latestRun: latestRun ? { runId: latestRun, passes: latest.filter(record => record.outcome === 'passed').length, trials: latest.length } : null,
});
}
const alarms: Alarm[] = [];
const rate = (stats: SeriesStats) => stats.passes / stats.trials;
for (const entry of cases) {
const current = entry.current;
if (!current) continue;
const below = current.trials >= policy.quarantine.entry.minTrials && rate(current) < policy.quarantine.entry.rate;
if (below && !entry.quarantined && BLOCKING_TIERS.includes(entry.tier)) alarms.push({ kind: 'drift', case: entry.case,
message: `${entry.case} passes ${current.passes}/${current.trials} (below ${pct(policy.quarantine.entry.rate)} over >= ${policy.quarantine.entry.minTrials} trials): fix it, or propose a CASE_QUARANTINE entry with a written diagnosis (product defects are never quarantined)` });
if (below && entry.kind === 'rule') alarms.push({ kind: 'rule-as-behavior', case: entry.case,
message: `${entry.case}: rule case behaving like behavior (${current.passes}/${current.trials}): fix or reclassify` });
if (entry.quarantined && current.trials >= policy.quarantine.exit.minTrials && rate(current) >= policy.quarantine.exit.rate) {
alarms.push({ kind: 'quarantine-exit', case: entry.case,
message: `${entry.case} passes ${current.passes}/${current.trials} (>= ${pct(policy.quarantine.exit.rate)}): remove its CASE_QUARANTINE entry` });
}
}
const tested = cases.filter(entry => BLOCKING_TIERS.includes(entry.tier) && entry.current && entry.previous
&& entry.current.trials >= policy.drift.fisherMinPerSide && entry.previous.trials >= policy.drift.fisherMinPerSide);
const pValues = tested.map(entry => fisherOneSidedLower(entry.current!.passes, entry.current!.trials, entry.previous!.passes, entry.previous!.trials));
for (const index of holmRejections(pValues, policy.drift.fisherAlpha)) {
const entry = tested[index]!;
alarms.push({ kind: 'regression', case: entry.case,
message: `${entry.case}: current identity ${entry.current!.passes}/${entry.current!.trials} is significantly below the previous ${entry.previous!.passes}/${entry.previous!.trials} (one-sided Fisher p=${pValues[index]!.toFixed(4)}, Holm over ${tested.length} cases)` });
}
for (const [id, entry] of Object.entries(quarantine)) {
const runs = quarantineRunsSince(entry.enteredAt, options.weeklyRuns, now);
if (runs >= policy.quarantine.expiryWeeklyRuns) alarms.push({ kind: 'quarantine-expired', case: id,
message: `${id}: entered ${runs} weekly runs ago (limit ${policy.quarantine.expiryWeeklyRuns}): fix it, name it as a red, or re-diagnose with fresh evidence` });
}
alarms.push(...quarantinePolicyProblems(quarantine, registry, policy, now));
const post = records.filter(record => record.policy_version !== 0).length;
return { policyVersion: policy.version, cases, alarms, postPolicyTrials: post, prePolicyTrials: records.length - post,
unattributed: options.unattributed ?? [], errors: options.errors ?? [] };
}
function pct(value: number): string { return `${Math.round(value * 1000) / 10}%`; }
function formatStats(stats: SeriesStats | null): string {
if (!stats) return '-';
return `${stats.passes}/${stats.trials} [${pct(stats.interval.lo)}–${pct(stats.interval.hi)}]${stats.infra ? ` (${stats.infra} infra)` : ''}`;
}
export function formatPassRates(report: PassRateReport, options: { caseFilter?: string } = {}): string {
const lines: string[] = [];
lines.push(`pass-rates: policy v${report.policyVersion}, ${report.postPolicyTrials} post-policy trial(s), ${report.prePolicyTrials} pre-policy (display only)`);
if (report.postPolicyTrials === 0) lines.push(' no post-policy trials yet: every series starts INCONCLUSIVE');
const cases = report.cases.filter(entry => !options.caseFilter || entry.case === options.caseFilter);
lines.push(' label kind tier current series pre-policy manual case');
for (const entry of cases) {
const group = entry.current ? ` ${entry.current.model} / ${entry.current.cli}` : '';
const reset = entry.previous ? ' (baseline reset)' : '';
lines.push(` ${entry.label.padEnd(12)} ${entry.kind.padEnd(8)} ${entry.tier.padEnd(8)} ${formatStats(entry.current).padEnd(29)} `
+ `${formatStats(entry.prePolicy).padEnd(18)} ${String(entry.manualReviews).padStart(6)} ${entry.case}${entry.quarantined ? ' [quarantined]' : ''}${group}${reset}`);
}
if (report.unattributed.length) lines.push(` unattributed records (${report.unattributed.length}, never guessed): ${report.unattributed.slice(0, 20).join(', ')}`);
if (report.errors.length) lines.push(` rejected ${report.errors.length} invalid trial line(s): ${report.errors.slice(0, 5).join('; ')}`);
if (report.alarms.length) {
lines.push(`ACTION REQUIRED (${report.alarms.length}):`);
for (const alarm of report.alarms) lines.push(` [${alarm.kind}] ${alarm.message}`);
}
return lines.join('\n');
}
// --- GitHub history ---
export interface WeeklyRun { id: number; attempt: number; sha: string; branch: string; createdAt: string }
export interface RunArtifact { id: number; name: string; size: number }
/** The GitHub calls pass-rates makes; injectable so the free tests never touch the network. */
export interface HistoryFetcher {
listRuns(repo: string, workflow: string, branch: string, limit: number): WeeklyRun[];
listArtifacts(repo: string, runId: number): RunArtifact[];
downloadZip(repo: string, artifactId: number, destination: string): void;
}
function gh(args: string[]): Buffer {
const result = spawnSync('gh', args, { timeout: 300_000, maxBuffer: 256 * 1024 * 1024 });
if (result.status !== 0) throw new Error(`gh ${args.slice(0, 2).join(' ')} failed: ${String(result.stderr || result.error || '').trim()}`);
return result.stdout;
}
const jsonLines = <T>(buffer: Buffer): T[] => buffer.toString('utf8').split('\n').filter(Boolean).map(line => JSON.parse(line) as T);
export const GH_HISTORY: HistoryFetcher = {
listRuns: (repo, workflow, branch, limit) => jsonLines<WeeklyRun>(gh(['api',
`repos/${repo}/actions/workflows/${workflow}/runs?branch=${encodeURIComponent(branch)}&status=completed&per_page=${limit}`,
'--jq', '.workflow_runs[] | {id, attempt: .run_attempt, sha: .head_sha, branch: .head_branch, createdAt: .created_at}'])),
listArtifacts: (repo, runId) => jsonLines<RunArtifact>(gh(['api', `repos/${repo}/actions/runs/${runId}/artifacts?per_page=100`,
'--paginate', '--jq', '.artifacts[] | select(.expired | not) | {id, name, size: .size_in_bytes}'])),
downloadZip: (repo, artifactId, destination) => fs.writeFileSync(destination, gh(['api', `repos/${repo}/actions/artifacts/${artifactId}/zip`])),
};
/** The last `limit` completed runs of `workflow` on each branch, newest first, deduplicated. */
export function listWeeklyRuns(opts: { repo: string; workflow: string; branches: string[]; limit: number; fetcher?: HistoryFetcher }): WeeklyRun[] {
const fetcher = opts.fetcher ?? GH_HISTORY;
const runs = new Map<number, WeeklyRun>();
for (const branch of opts.branches) for (const run of fetcher.listRuns(opts.repo, opts.workflow, branch, opts.limit)) runs.set(run.id, run);
return [...runs.values()].sort((a, b) => b.createdAt.localeCompare(a.createdAt));
}
/**
* Download the artifacts of one run whose names match into a per-run cache
* directory (reused on later calls) and return the extracted directories.
* Oversized or oddly named artifacts are skipped: downloads are data only.
*/
export function downloadRunArtifacts(opts: { repo: string; run: WeeklyRun; match: (name: string) => boolean; cacheDir: string;
fetcher?: HistoryFetcher; maxBytes?: number }): string[] {
const fetcher = opts.fetcher ?? GH_HISTORY;
const dirs: string[] = [];
for (const artifact of fetcher.listArtifacts(opts.repo, opts.run.id)) {
if (!opts.match(artifact.name) || !/^[A-Za-z0-9._-]+$/.test(artifact.name)) continue;
if (artifact.size > (opts.maxBytes ?? TRIAL_OUTCOMES_MAX_BYTES)) continue;
const dir = path.join(opts.cacheDir, `${opts.run.id}`, artifact.name);
if (!fs.existsSync(path.join(dir, '.complete'))) {
fs.rmSync(dir, { recursive: true, force: true });
fs.mkdirSync(dir, { recursive: true });
const zip = path.join(dir, 'artifact.zip');
fetcher.downloadZip(opts.repo, artifact.id, zip);
const unzip = spawnSync('unzip', ['-o', '-q', zip, '-d', dir], { timeout: 120_000 });
if (unzip.status !== 0) throw new Error(`unzip failed for ${artifact.name}: ${String(unzip.stderr || unzip.error || '')}`);
fs.rmSync(zip, { force: true });
fs.writeFileSync(path.join(dir, '.complete'), '');
}
dirs.push(dir);
}
return dirs;
}
function gitOutput(args: string[]): string | null {
const result = spawnSync('git', args, { encoding: 'utf8', timeout: 5_000 });
return result.status === 0 ? result.stdout.trim() : null;
}
function repoSlug(): string {
const url = gitOutput(['remote', 'get-url', 'origin']) ?? '';
return url.match(/[:/]([^/:]+\/[^/]+?)(?:\.git)?$/)?.[1] ?? 'garrytan/gstack';
}
if (import.meta.main) {
const argv = process.argv.slice(2);
const flag = (name: string) => { const index = argv.indexOf(name); return index === -1 ? undefined : argv[index + 1]; };
const dirs = argv.flatMap((arg, index) => arg === '--dir' && argv[index + 1] ? [argv[index + 1]!] : []);
const asJson = argv.includes('--json');
const gate = argv.includes('--gate');
const backfill = argv.includes('--backfill');
const caseFilter = flag('--case');
const runsLimit = Number(flag('--runs')) || 10;
const sinceDays = Number(flag('--since-days')) || 60;
const repo = flag('--repo') ?? repoSlug();
const workflow = flag('--workflow') ?? 'evals-periodic.yml';
const branch = flag('--branch') ?? gitOutput(['rev-parse', '--abbrev-ref', 'HEAD']) ?? 'main';
const records: TrialRecord[] = [];
const unattributed = new Set<string>();
const errors: string[] = [];
let historyError: string | null = null;
let weeklyRuns: string[] | undefined;
const manualReviews: string[] = [];
const importDir = (dir: string, run: { run_id: string; sha?: string; timestamp?: string } | undefined, legacyDays: number) => {
const trials = readTrialOutcomeDir(dir);
records.push(...trials.records);
errors.push(...trials.errors);
const legacy = backfillEvalFiles(collectEvalFiles(dir, legacyDays), run);
records.push(...legacy.records);
manualReviews.push(...legacy.manualReviews);
legacy.unattributed.forEach(name => unattributed.add(name));
};
if (dirs.length) {
for (const dir of dirs) importDir(dir, undefined, sinceDays);
} else {
try {
const runs = listWeeklyRuns({ repo, workflow, branches: [...new Set([branch, 'main'])], limit: runsLimit });
weeklyRuns = runs.map(run => run.createdAt);
const cacheDir = path.join(os.homedir(), '.gstack', 'eval-pass-rates-cache', repo.replace('/', '-'));
const match = backfill
? (name: string) => name.startsWith('trial-outcomes') || /^(paid-slice-\d+|gate-census-\d+)$/.test(name)
: (name: string) => name.startsWith('trial-outcomes');
for (const run of runs) {
const dirsForRun = downloadRunArtifacts({ repo, run, match, cacheDir, maxBytes: backfill ? 64 * 1024 * 1024 : undefined });
for (const dir of dirsForRun) importDir(dir, { run_id: `${run.id}`, sha: run.sha, timestamp: run.createdAt }, 3650);
}
} catch (error) {
historyError = error instanceof Error ? error.message : String(error);
}
}
const report = analyzePassRates(records, { weeklyRuns, unattributed: [...unattributed].sort(), errors, manualReviews });
const ledger = readFreeLedger();
if (asJson) {
console.log(JSON.stringify({ repo, workflow, branch, dirs, historyError, ...report, freeLedger: ledger }, null, 2));
} else {
if (historyError) console.log(`pass-rates: history unavailable (${historyError}); every label below is INCONCLUSIVE`);
console.log(formatPassRates(report, { caseFilter }));
if (ledger.length > 0) {
const byFile = new Map<string, number>();
for (const e of ledger) byFile.set(e.file, (byFile.get(e.file) ?? 0) + 1);
console.log(`free-suite flaky-passes (${flakeLedgerPath()}):`);
for (const [file, n] of [...byFile.entries()].sort((a, b) => b[1] - a[1])) {
console.log(` ${String(n).padStart(3)}x ${file}`);
}
}
}
if (gate && (historyError || report.alarms.length)) process.exit(1);
}