feat(evals): per-case pass rates with Wilson intervals, identity series and quarantine policy

scripts/eval-flake-rank.ts becomes eval:pass-rates (eval:flake-rank stays an
alias, and the legacy aggregate stays exported). It reads eval-store's
trial-outcomes JSONL from the last N completed evals-periodic runs on this
branch and main (gh, downloading only the trial-outcomes artifact, cached and
size-capped, parsed as data), plus local eval dirs, and prints per-case
per-trial pass rates with 95% Wilson intervals.

A series is a case's own touchfiles minus GLOBAL_TOUCHFILES
(caseSeriesIdentities, for the report job to stamp), per model, CLI version
and policy version. Labels: INCONCLUSIVE, BROKEN, FLAKY, FAILING, PASSING.
--backfill imports legacy slice artifacts as pre-policy trials (first
attempt only, attributed by registry id, never guessed) for display only.

--gate fails with ACTION REQUIRED on post-policy evidence only: drift below
the quarantine entry rule, a rule case behaving like behavior, a one-sided
Fisher drop against the previous identity (Holm-controlled), and quarantine
entries that met their exit rule, expired after 8 weekly runs, broke the
10% tier cap or are invalid. CASE_QUARANTINE entries now carry a
failureClass (detector, harness or model-latency); a product defect has no
class and is never quarantined. The policy test pins EVAL_POLICY's approved
constants.
This commit is contained in:
garrytan committed 2026-09-29 19:11:27 +00:00
1 parent 0292ee3f6f
commit d6b3559b78
4 files changed
+931 -44

No files matched your search

+586 -34
View File
@@ -1,29 +1,55 @@
#!/usr/bin/env bun
/**
* eval-flake-rank — the flake-telemetry dial (WS1).
* eval-pass-rates (alias: eval-flake-rank) — per-case trial pass rates.
*
* Aggregates per-test series across every FINALIZED eval-store run on this
* machine (default: ~/.gstack/projects/<slug>/evals/, shard dirs included)
* plus the free suite's flake ledger, and ranks tests by flake signal:
* retried passes first (a test that needs attempt 2 to go green is the
* definition of a flake), then failure rate.
* Reads trial records (one JSONL line per trial: case, kind, trial, outcome,
* exit_reason, duration, cost, model, CLI version, series identity, run id,
* sha, policy_version) from the last N completed `evals-periodic.yml` runs on
* the current branch and `main` (downloading only each run's small
* `trial-outcomes` artifact through `gh`), plus any local eval dirs, and
* prints per-case per-trial pass rates with 95% Wilson intervals.
*
* This is the readable dial behind two policies:
* - a flaky pass never blocks a merge, but it is recorded and RANKED here;
* - the required-check promotion (WS16) needs weeks of clean flake-rank,
* not vibes.
* A series is one case under one input identity: the case's own touchfiles
* minus GLOBAL_TOUCHFILES (`caseSeriesIdentities`), grouped by model and CLI
* version, per policy_version. A new identity starts a new series; earlier
* series stay visible. Only post-policy trials of the current series feed the
* labels and alarms. Legacy eval-store records (`--backfill`, `--dir`) are
* imported as pre-policy trials (first attempt only; a missing attempt means
* 1) and are display-only.
*
* Labels: INCONCLUSIVE (below the entry rule's minimum trials), BROKEN (latest run 0/n
* after a prior interval at or above the entry rate), FLAKY (failures and an
* interval straddling the entry rate), FAILING (interval below the entry
* rate), PASSING (otherwise).
*
* The weekly gate (`--gate`) exits non-zero with ACTION REQUIRED when a
* non-quarantined case meets the quarantine entry rule, a rule case behaves
* like a behavior case, a blocking case's current-identity rate is
* significantly below its previous identity (one-sided Fisher exact,
* Holm-controlled across cases), or a CASE_QUARANTINE entry has met its exit
* rule, expired, or pushed its tier over the cap. History that cannot be
* fetched fails the gate closed.
*
* Usage:
* bun run eval:flake-rank # project eval dir
* bun run eval:flake-rank --dir <path> # e.g. downloaded CI artifacts
* bun run eval:flake-rank --json # machine-readable
* bun run eval:pass-rates # last 10 weekly runs, this branch + main
* bun run eval:pass-rates --case <id> --runs 20
* bun run eval:pass-rates --dir <path> # local eval dirs / downloaded artifacts (repeatable)
* bun run eval:pass-rates --backfill # also import legacy slice artifacts, labeled pre-policy
* bun run eval:pass-rates --json | --gate
*/
import * as fs from 'node:fs';
import * as os from 'node:os';
import * as path from 'node:path';
import { getProjectEvalDir, isPartialEval, isFinalizedEvalResultFile, type EvalResult } from '../test/helpers/eval-store';
import { evalEntryOutcome } from '../test/helpers/eval-store';
import { spawnSync } from 'node:child_process';
import { createHash } from 'node:crypto';
import { isPartialEval, isFinalizedEvalResultFile, evalEntryOutcome, failureClassOf, parseTrialOutcomes, sanitizeTrialError,
TRIAL_OUTCOME_SCHEMA, type EvalCaseKind, type EvalResult, type TrialOutcomeRecord } from '../test/helpers/eval-store';
import { flakeLedgerPath, type FlakeLedgerEntry } from './test-free-shards';
import { E2E_KINDS, E2E_TIERS, E2E_TOUCHFILES, GLOBAL_TOUCHFILES, LLM_JUDGE_TOUCHFILES } from '../test/helpers/touchfiles-data';
import { CASE_QUARANTINE, EVAL_POLICY } from '../test/helpers/periodic-exclude-data';
import { matchGlob } from '../test/helpers/test-selection';
import { CASE_TEST_NAMES } from './test-paid-shards';
interface TestSeries {
name: string;
@@ -111,33 +137,558 @@ function readFreeLedger(): FlakeLedgerEntry[] {
return out;
}
// --- Trial records ---
/**
* A trial record as pass-rates reads it: eval-store's trial-outcomes schema
* plus the series identity the report job stamps (caseSeriesIdentities).
* policy_version 0 marks a pre-policy (backfilled) record.
*/
export type TrialRecord = TrialOutcomeRecord & { series_identity?: string };
/** Per-file cap for downloaded artifacts: pass-rates parses data only, never executes it. */
export const TRIAL_OUTCOMES_MAX_BYTES = 8 * 1024 * 1024;
/** Every `trial-outcomes*.jsonl` file under a directory, size-capped, schema-validated by eval-store. */
export function readTrialOutcomeDir(dir: string): { records: TrialRecord[]; errors: string[] } {
const records: TrialRecord[] = [];
const errors: string[] = [];
if (!fs.existsSync(dir)) return { records, errors };
for (const name of fs.readdirSync(dir, { recursive: true }) as string[]) {
if (!/(^|\/)trial-outcomes[^/]*\.jsonl$/.test(name)) continue;
const full = path.join(dir, name);
const parsed = parseTrialOutcomes(fs.readFileSync(full, 'utf8'), { maxBytes: TRIAL_OUTCOMES_MAX_BYTES });
records.push(...parsed.records.map(record => ({
...record, series_identity: typeof (record as TrialRecord).series_identity === 'string'
? (record as TrialRecord).series_identity!.slice(0, 64) : undefined })));
errors.push(...parsed.errors.map(error => `${full}: ${error}`));
}
return { records, errors };
}
// --- Registry attribution and series identity ---
export interface Registry {
kinds: Record<string, EvalCaseKind>;
tiers: Record<string, string>;
touchfiles: Record<string, string[]>;
judgeTouchfiles: Record<string, string[]>;
globals: readonly string[];
testNames: Record<string, string>;
}
export const LIVE_REGISTRY: Registry = {
kinds: E2E_KINDS, tiers: E2E_TIERS, touchfiles: E2E_TOUCHFILES, judgeTouchfiles: LLM_JUDGE_TOUCHFILES,
globals: GLOBAL_TOUCHFILES, testNames: CASE_TEST_NAMES,
};
/** A case's tier: its E2E_TIERS value, or 'judge' for an LLM-judge entry. */
export function caseTier(id: string, registry: Registry = LIVE_REGISTRY): string {
return registry.tiers[id] ?? (id in registry.judgeTouchfiles ? 'judge' : 'unknown');
}
/**
* Attribute a legacy eval-store record to a registry id: the case-shard slug
* suffix (`<file>--<id>`), the recorded name, a CASE_TEST_NAMES label, or the
* only id its shard file registers. Anything else is unattributed (null).
*/
export function attributeLegacyRecord(name: string, shard: string | undefined, registry: Registry = LIVE_REGISTRY): string | null {
const known = (id: string) => id in registry.kinds;
const [slugFile, slugCase] = (shard ?? '').split('--');
if (slugCase && known(slugCase)) return slugCase;
if (known(name)) return name;
const labeled = Object.entries(registry.testNames).find(([, label]) => label === name)?.[0];
if (labeled && known(labeled)) return labeled;
if (slugFile) {
const file = `test/${slugFile}.test.ts`;
const owners = Object.keys(registry.touchfiles).filter(id => registry.touchfiles[id]!.includes(file));
if (owners.length === 1 && known(owners[0]!)) return owners[0]!;
}
return null;
}
/**
* Series identity per case: a hash of the git blob ids of the files matching
* the case's own touchfiles, excluding GLOBAL_TOUCHFILES (harness edits are
* markers, not new series). The report job stamps this on every trial record.
*/
export function caseSeriesIdentities(ids: string[], root: string, registry: Registry = LIVE_REGISTRY): Record<string, string> {
const listed = spawnSync('git', ['ls-files', '-s'], { cwd: root, encoding: 'utf8', timeout: 20_000, maxBuffer: 64 * 1024 * 1024 });
if (listed.status !== 0) throw new Error(`git ls-files failed: ${listed.stderr}`);
const blobs = listed.stdout.split('\n').filter(Boolean).map(line => {
const [meta, file] = line.split('\t');
return { file: file!, blob: meta!.split(' ')[1]! };
}).filter(entry => !registry.globals.some(pattern => matchGlob(entry.file, pattern)));
return Object.fromEntries(ids.map(id => {
const patterns = registry.touchfiles[id] ?? registry.judgeTouchfiles[id] ?? [];
const lines = blobs.filter(entry => patterns.some(pattern => matchGlob(entry.file, pattern)))
.map(entry => `${entry.file} ${entry.blob}`).sort();
return [id, createHash('sha256').update(`${id}\n${lines.join('\n')}`).digest('hex').slice(0, 16)];
}));
}
/**
* Import legacy eval-store result files as pre-policy trials (policy_version
* 0, source 'backfill'): first attempt only (a missing attempt means 1),
* attributed by registry id, never guessed. A manual-review acceptance carries
* no automated verdict: it is counted and shown, never scored. Without a CI
* run, each local result file is its own run.
*/
export function backfillEvalFiles(files: string[], run?: { run_id: string; sha?: string; timestamp?: string },
registry: Registry = LIVE_REGISTRY): { records: TrialRecord[]; unattributed: string[]; manualReviews: string[] } {
const records: TrialRecord[] = [];
const unattributed = new Set<string>();
const manualReviews: string[] = [];
for (const file of files) {
let result: EvalResult & { shard?: string; claude_cli_version?: string };
try { result = JSON.parse(fs.readFileSync(file, 'utf8')); } catch { continue; }
if (isPartialEval(result, file) || !Array.isArray(result.tests)) continue;
const seen = new Set<string>();
for (const entry of result.tests) {
if ((entry.attempt ?? 1) !== 1 || seen.has(entry.name)) continue;
seen.add(entry.name);
const id = attributeLegacyRecord(entry.name, result.shard, registry);
if (!id) { unattributed.add(entry.name); continue; }
const outcome = evalEntryOutcome(entry);
if (outcome === 'manual-review') { manualReviews.push(id); continue; }
records.push({
schema: TRIAL_OUTCOME_SCHEMA, case: id,
file: result.shard ? `test/${result.shard.split('--')[0]}.test.ts` : 'unknown',
tier: caseTier(id, registry), kind: registry.kinds[id]!, trial: 1, panel: { n: 1, k: 1 }, attempt: 1,
outcome, ...(outcome === 'failed' ? { failure_class: failureClassOf(entry) } : {}),
exit_reason: entry.exit_reason, error: sanitizeTrialError(entry.error),
duration_ms: Math.max(0, entry.duration_ms || 0), cost_usd: Math.max(0, entry.cost_usd || 0),
model: entry.model, cli_version: result.claude_cli_version, policy_version: 0, quarantined: false,
execution: entry.execution === 'reused' ? 'reused' : 'executed', source: 'backfill',
run_id: run?.run_id ?? `local:${file}`, sha: run?.sha ?? result.git_sha, recorded_at: run?.timestamp ?? result.timestamp,
});
}
}
return { records, unattributed: [...unattributed].sort(), manualReviews };
}
// --- Statistics ---
/** 95% Wilson score interval for k successes in n trials. */
export function wilsonInterval(k: number, n: number, z = 1.96): { lo: number; hi: number } {
if (n <= 0) return { lo: 0, hi: 1 };
const p = k / n, z2 = z * z, denom = 1 + z2 / n;
const center = (p + z2 / (2 * n)) / denom;
const half = (z * Math.sqrt(p * (1 - p) / n + z2 / (4 * n * n))) / denom;
return { lo: Math.max(0, center - half), hi: k === n ? 1 : Math.min(1, center + half) };
}
function logChoose(n: number, k: number): number {
let sum = 0;
for (let i = 1; i <= k; i++) sum += Math.log(n - k + i) - Math.log(i);
return sum;
}
/**
* One-sided Fisher exact p-value that the CURRENT pass rate is below the
* PREVIOUS one: P(X <= curPass) under the hypergeometric null with the
* observed margins.
*/
export function fisherOneSidedLower(curPass: number, curN: number, prevPass: number, prevN: number): number {
const passes = curPass + prevPass, total = curN + prevN;
const denom = logChoose(total, passes);
let p = 0;
for (let x = Math.max(0, passes - prevN); x <= curPass; x++) p += Math.exp(logChoose(curN, x) + logChoose(prevN, passes - x) - denom);
return Math.min(1, p);
}
/** Holm step-down: the indices whose p-values are rejected at family-wise alpha. */
export function holmRejections(pValues: number[], alpha: number): Set<number> {
const order = pValues.map((p, index) => ({ p, index })).sort((a, b) => a.p - b.p);
const rejected = new Set<number>();
for (let rank = 0; rank < order.length; rank++) {
if (order[rank]!.p > alpha / (order.length - rank)) break;
rejected.add(order[rank]!.index);
}
return rejected;
}
// --- Analysis ---
export type PassRateLabel = 'INCONCLUSIVE' | 'BROKEN' | 'FLAKY' | 'FAILING' | 'PASSING';
export type AlarmKind = 'drift' | 'rule-as-behavior' | 'regression' | 'quarantine-exit' | 'quarantine-expired'
| 'quarantine-cap' | 'quarantine-invalid';
/** The EVAL_POLICY fields pass-rates reads (structural, so tests can vary them). */
export interface PassRatePolicy {
version: number;
quarantine: { entry: { rate: number; minTrials: number }; exit: { rate: number; minTrials: number }; capFraction: number; expiryWeeklyRuns: number };
drift: { fisherAlpha: number; fisherMinPerSide: number };
}
export type QuarantineEntry = (typeof CASE_QUARANTINE)[string];
/** Tiers whose cases block a lane; quarantine applies only to them. */
export const BLOCKING_TIERS: readonly string[] = ['gate', 'periodic'];
const QUARANTINE_FAILURE_CLASSES: readonly string[] = ['detector', 'harness', 'model-latency'];
export interface SeriesStats {
key: string;
identity: string;
model: string;
cli: string;
policyVersion: number;
passes: number;
/** Scored trials: passed + failed (skipped trials carry no verdict). */
trials: number;
infra: number;
interval: { lo: number; hi: number };
firstSeen: string;
lastSeen: string;
runs: string[];
}
export interface CasePassRate {
case: string;
kind: EvalCaseKind;
tier: string;
quarantined: boolean;
label: PassRateLabel;
/** Manual-review acceptances: visible, never scored. */
manualReviews: number;
current: SeriesStats | null;
previous: SeriesStats | null;
prePolicy: SeriesStats | null;
series: SeriesStats[];
latestRun: { runId: string; passes: number; trials: number } | null;
}
export interface Alarm { kind: AlarmKind; case: string; message: string }
export interface PassRateReport {
policyVersion: number;
cases: CasePassRate[];
alarms: Alarm[];
postPolicyTrials: number;
prePolicyTrials: number;
unattributed: string[];
errors: string[];
}
export interface AnalyzeOptions {
registry?: Registry;
quarantine?: Record<string, QuarantineEntry>;
policy?: PassRatePolicy;
/** Completed weekly-run timestamps in the window, for quarantine expiry. */
weeklyRuns?: string[];
now?: number;
unattributed?: string[];
errors?: string[];
manualReviews?: string[];
}
const at = (record: TrialRecord) => record.recorded_at ?? '';
const runOf = (record: TrialRecord) => `${record.run_id ?? record.sha ?? 'local'}#${record.attempt}`;
function seriesStats(key: string, records: TrialRecord[]): SeriesStats {
const scored = records.filter(record => record.outcome !== 'skipped');
const passes = scored.filter(record => record.outcome === 'passed').length;
const times = records.map(at).sort();
const first = records[0]!;
return {
key, identity: first.series_identity ?? 'unknown', model: first.model ?? 'unknown', cli: first.cli_version ?? 'unknown',
policyVersion: first.policy_version, passes, trials: scored.length,
infra: scored.filter(record => record.outcome === 'failed' && record.failure_class === 'infra').length,
interval: wilsonInterval(passes, scored.length), firstSeen: times[0] ?? '', lastSeen: times[times.length - 1] ?? '',
runs: [...new Set(records.map(runOf))],
};
}
/** Weekly runs completed after an entry's enteredAt; offline, whole weeks elapsed. */
export function quarantineRunsSince(enteredAt: string, weeklyRuns: string[] | undefined, now: number): number {
const entered = Date.parse(enteredAt);
if (!Number.isFinite(entered)) return Number.POSITIVE_INFINITY;
if (weeklyRuns && weeklyRuns.length) return weeklyRuns.filter(time => Date.parse(time) > entered).length;
return Math.floor((now - entered) / (7 * 86_400_000));
}
/**
* Static CASE_QUARANTINE problems, shared by the free policy test and the
* weekly gate: an id that is not a blocking-tier E2E case, a missing field,
* a failure class outside detector / harness / model-latency (a product
* defect is fixed or named, never quarantined), a malformed or future date,
* and a tier over its cap.
*/
export function quarantinePolicyProblems(quarantine: Record<string, QuarantineEntry>,
registry: Registry = LIVE_REGISTRY, policy: PassRatePolicy = EVAL_POLICY, now = Date.now()): Alarm[] {
const problems: Alarm[] = [];
const invalid = (id: string, message: string) => problems.push({ kind: 'quarantine-invalid', case: id, message: `${id}: ${message}` });
const perTier = new Map<string, number>();
for (const [id, entry] of Object.entries(quarantine)) {
const tier = registry.tiers[id];
if (!tier || !(id in registry.kinds)) { invalid(id, 'CASE_QUARANTINE names no registered E2E case'); continue; }
if (!BLOCKING_TIERS.includes(tier)) invalid(id, `tier ${tier} is not blocking; only ${BLOCKING_TIERS.join(' and ')} cases are quarantined`);
for (const field of ['reason', 'failureClass', 'tracking', 'owner', 'enteredAt', 'exit'] as const) {
if (typeof entry[field] !== 'string' || !entry[field].trim()) invalid(id, `missing ${field}`);
}
if (typeof entry.reason === 'string' && entry.reason.trim().length < 40) invalid(id, 'reason must be a written diagnosis (at least 40 characters)');
if (!QUARANTINE_FAILURE_CLASSES.includes(entry.failureClass)) {
invalid(id, `failureClass ${JSON.stringify(entry.failureClass)} is not ${QUARANTINE_FAILURE_CLASSES.join(', ')}; a product defect is fixed or named as a red, never quarantined`);
}
const entered = Date.parse(entry.enteredAt);
if (!/^\d{4}-\d{2}-\d{2}$/.test(entry.enteredAt ?? '') || !Number.isFinite(entered)) invalid(id, 'enteredAt must be YYYY-MM-DD');
else if (entered > now) invalid(id, 'enteredAt is in the future');
perTier.set(tier, (perTier.get(tier) ?? 0) + 1);
}
for (const [tier, count] of perTier) {
const size = Object.values(registry.tiers).filter(value => value === tier).length;
const cap = Math.floor(size * policy.quarantine.capFraction);
if (count > cap) problems.push({ kind: 'quarantine-cap', case: tier,
message: `${count} quarantined ${tier} cases exceed the ${pct(policy.quarantine.capFraction)} cap (${cap} of ${size})` });
}
return problems;
}
export function analyzePassRates(records: TrialRecord[], options: AnalyzeOptions = {}): PassRateReport {
const registry = options.registry ?? LIVE_REGISTRY;
const quarantine = options.quarantine ?? CASE_QUARANTINE;
const policy = options.policy ?? EVAL_POLICY;
const now = options.now ?? Date.now();
const byCase = new Map<string, TrialRecord[]>();
for (const record of records) {
const list = byCase.get(record.case) ?? [];
list.push(record);
byCase.set(record.case, list);
}
for (const id of options.manualReviews ?? []) if (!byCase.has(id)) byCase.set(id, []);
const cases: CasePassRate[] = [];
for (const [id, list] of [...byCase].sort(([a], [b]) => a.localeCompare(b))) {
list.sort((a, b) => at(a).localeCompare(at(b)) || runOf(a).localeCompare(runOf(b)) || a.trial - b.trial);
const groups = new Map<string, TrialRecord[]>();
for (const record of list) {
const key = record.policy_version === 0 ? 'pre-policy'
: [record.series_identity ?? 'unknown', record.model ?? 'unknown', record.cli_version ?? 'unknown', `v${record.policy_version}`].join('|');
const group = groups.get(key) ?? [];
group.push(record);
groups.set(key, group);
}
const series = [...groups].map(([key, group]) => seriesStats(key, group))
.sort((a, b) => a.lastSeen.localeCompare(b.lastSeen));
const post = series.filter(entry => entry.policyVersion !== 0);
const current = post[post.length - 1] ?? null;
const previous = post[post.length - 2] ?? null;
const scored = current ? groups.get(current.key)!.filter(record => record.outcome !== 'skipped') : [];
const latestRun = scored.length ? runOf(scored[scored.length - 1]!) : null;
const latest = scored.filter(record => runOf(record) === latestRun);
const prior = scored.filter(record => runOf(record) !== latestRun);
const priorPasses = prior.filter(record => record.outcome === 'passed').length;
const entryRate = policy.quarantine.entry.rate;
let label: PassRateLabel;
if (latest.length > 0 && latest.every(record => record.outcome === 'failed')
&& prior.length > 0 && wilsonInterval(priorPasses, prior.length).lo >= entryRate) label = 'BROKEN';
else if (!current || current.trials < policy.quarantine.entry.minTrials) label = 'INCONCLUSIVE';
else if (current.interval.hi < entryRate) label = 'FAILING';
else if (current.passes < current.trials && current.interval.lo < entryRate) label = 'FLAKY';
else label = 'PASSING';
cases.push({
case: id, kind: registry.kinds[id] ?? list[0]!.kind, tier: caseTier(id, registry),
quarantined: id in quarantine, label, current, previous,
manualReviews: (options.manualReviews ?? []).filter(name => name === id).length,
prePolicy: series.find(entry => entry.policyVersion === 0) ?? null, series,
latestRun: latestRun ? { runId: latestRun, passes: latest.filter(record => record.outcome === 'passed').length, trials: latest.length } : null,
});
}
const alarms: Alarm[] = [];
const rate = (stats: SeriesStats) => stats.passes / stats.trials;
for (const entry of cases) {
const current = entry.current;
if (!current) continue;
const below = current.trials >= policy.quarantine.entry.minTrials && rate(current) < policy.quarantine.entry.rate;
if (below && !entry.quarantined && BLOCKING_TIERS.includes(entry.tier)) alarms.push({ kind: 'drift', case: entry.case,
message: `${entry.case} passes ${current.passes}/${current.trials} (below ${pct(policy.quarantine.entry.rate)} over >= ${policy.quarantine.entry.minTrials} trials): fix it, or propose a CASE_QUARANTINE entry with a written diagnosis (product defects are never quarantined)` });
if (below && entry.kind === 'rule') alarms.push({ kind: 'rule-as-behavior', case: entry.case,
message: `${entry.case}: rule case behaving like behavior (${current.passes}/${current.trials}): fix or reclassify` });
if (entry.quarantined && current.trials >= policy.quarantine.exit.minTrials && rate(current) >= policy.quarantine.exit.rate) {
alarms.push({ kind: 'quarantine-exit', case: entry.case,
message: `${entry.case} passes ${current.passes}/${current.trials} (>= ${pct(policy.quarantine.exit.rate)}): remove its CASE_QUARANTINE entry` });
}
}
const tested = cases.filter(entry => BLOCKING_TIERS.includes(entry.tier) && entry.current && entry.previous
&& entry.current.trials >= policy.drift.fisherMinPerSide && entry.previous.trials >= policy.drift.fisherMinPerSide);
const pValues = tested.map(entry => fisherOneSidedLower(entry.current!.passes, entry.current!.trials, entry.previous!.passes, entry.previous!.trials));
for (const index of holmRejections(pValues, policy.drift.fisherAlpha)) {
const entry = tested[index]!;
alarms.push({ kind: 'regression', case: entry.case,
message: `${entry.case}: current identity ${entry.current!.passes}/${entry.current!.trials} is significantly below the previous ${entry.previous!.passes}/${entry.previous!.trials} (one-sided Fisher p=${pValues[index]!.toFixed(4)}, Holm over ${tested.length} cases)` });
}
for (const [id, entry] of Object.entries(quarantine)) {
const runs = quarantineRunsSince(entry.enteredAt, options.weeklyRuns, now);
if (runs >= policy.quarantine.expiryWeeklyRuns) alarms.push({ kind: 'quarantine-expired', case: id,
message: `${id}: entered ${runs} weekly runs ago (limit ${policy.quarantine.expiryWeeklyRuns}): fix it, name it as a red, or re-diagnose with fresh evidence` });
}
alarms.push(...quarantinePolicyProblems(quarantine, registry, policy, now));
const post = records.filter(record => record.policy_version !== 0).length;
return { policyVersion: policy.version, cases, alarms, postPolicyTrials: post, prePolicyTrials: records.length - post,
unattributed: options.unattributed ?? [], errors: options.errors ?? [] };
}
function pct(value: number): string { return `${Math.round(value * 1000) / 10}%`; }
function formatStats(stats: SeriesStats | null): string {
if (!stats) return '-';
return `${stats.passes}/${stats.trials} [${pct(stats.interval.lo)}–${pct(stats.interval.hi)}]${stats.infra ? ` (${stats.infra} infra)` : ''}`;
}
export function formatPassRates(report: PassRateReport, options: { caseFilter?: string } = {}): string {
const lines: string[] = [];
lines.push(`pass-rates: policy v${report.policyVersion}, ${report.postPolicyTrials} post-policy trial(s), ${report.prePolicyTrials} pre-policy (display only)`);
if (report.postPolicyTrials === 0) lines.push(' no post-policy trials yet: every series starts INCONCLUSIVE');
const cases = report.cases.filter(entry => !options.caseFilter || entry.case === options.caseFilter);
lines.push(' label kind tier current series pre-policy manual case');
for (const entry of cases) {
const group = entry.current ? ` ${entry.current.model} / ${entry.current.cli}` : '';
const reset = entry.previous ? ' (baseline reset)' : '';
lines.push(` ${entry.label.padEnd(12)} ${entry.kind.padEnd(8)} ${entry.tier.padEnd(8)} ${formatStats(entry.current).padEnd(29)} `
+ `${formatStats(entry.prePolicy).padEnd(18)} ${String(entry.manualReviews).padStart(6)} ${entry.case}${entry.quarantined ? ' [quarantined]' : ''}${group}${reset}`);
}
if (report.unattributed.length) lines.push(` unattributed records (${report.unattributed.length}, never guessed): ${report.unattributed.slice(0, 20).join(', ')}`);
if (report.errors.length) lines.push(` rejected ${report.errors.length} invalid trial line(s): ${report.errors.slice(0, 5).join('; ')}`);
if (report.alarms.length) {
lines.push(`ACTION REQUIRED (${report.alarms.length}):`);
for (const alarm of report.alarms) lines.push(` [${alarm.kind}] ${alarm.message}`);
}
return lines.join('\n');
}
// --- GitHub history ---
export interface WeeklyRun { id: number; attempt: number; sha: string; branch: string; createdAt: string }
export interface RunArtifact { id: number; name: string; size: number }
/** The GitHub calls pass-rates makes; injectable so the free tests never touch the network. */
export interface HistoryFetcher {
listRuns(repo: string, workflow: string, branch: string, limit: number): WeeklyRun[];
listArtifacts(repo: string, runId: number): RunArtifact[];
downloadZip(repo: string, artifactId: number, destination: string): void;
}
function gh(args: string[]): Buffer {
const result = spawnSync('gh', args, { timeout: 300_000, maxBuffer: 256 * 1024 * 1024 });
if (result.status !== 0) throw new Error(`gh ${args.slice(0, 2).join(' ')} failed: ${String(result.stderr || result.error || '').trim()}`);
return result.stdout;
}
const jsonLines = <T>(buffer: Buffer): T[] => buffer.toString('utf8').split('\n').filter(Boolean).map(line => JSON.parse(line) as T);
export const GH_HISTORY: HistoryFetcher = {
listRuns: (repo, workflow, branch, limit) => jsonLines<WeeklyRun>(gh(['api',
`repos/${repo}/actions/workflows/${workflow}/runs?branch=${encodeURIComponent(branch)}&status=completed&per_page=${limit}`,
'--jq', '.workflow_runs[] | {id, attempt: .run_attempt, sha: .head_sha, branch: .head_branch, createdAt: .created_at}'])),
listArtifacts: (repo, runId) => jsonLines<RunArtifact>(gh(['api', `repos/${repo}/actions/runs/${runId}/artifacts?per_page=100`,
'--paginate', '--jq', '.artifacts[] | select(.expired | not) | {id, name, size: .size_in_bytes}'])),
downloadZip: (repo, artifactId, destination) => fs.writeFileSync(destination, gh(['api', `repos/${repo}/actions/artifacts/${artifactId}/zip`])),
};
/** The last `limit` completed runs of `workflow` on each branch, newest first, deduplicated. */
export function listWeeklyRuns(opts: { repo: string; workflow: string; branches: string[]; limit: number; fetcher?: HistoryFetcher }): WeeklyRun[] {
const fetcher = opts.fetcher ?? GH_HISTORY;
const runs = new Map<number, WeeklyRun>();
for (const branch of opts.branches) for (const run of fetcher.listRuns(opts.repo, opts.workflow, branch, opts.limit)) runs.set(run.id, run);
return [...runs.values()].sort((a, b) => b.createdAt.localeCompare(a.createdAt));
}
/**
* Download the artifacts of one run whose names match into a per-run cache
* directory (reused on later calls) and return the extracted directories.
* Oversized or oddly named artifacts are skipped: downloads are data only.
*/
export function downloadRunArtifacts(opts: { repo: string; run: WeeklyRun; match: (name: string) => boolean; cacheDir: string;
fetcher?: HistoryFetcher; maxBytes?: number }): string[] {
const fetcher = opts.fetcher ?? GH_HISTORY;
const dirs: string[] = [];
for (const artifact of fetcher.listArtifacts(opts.repo, opts.run.id)) {
if (!opts.match(artifact.name) || !/^[A-Za-z0-9._-]+$/.test(artifact.name)) continue;
if (artifact.size > (opts.maxBytes ?? TRIAL_OUTCOMES_MAX_BYTES)) continue;
const dir = path.join(opts.cacheDir, `${opts.run.id}`, artifact.name);
if (!fs.existsSync(path.join(dir, '.complete'))) {
fs.rmSync(dir, { recursive: true, force: true });
fs.mkdirSync(dir, { recursive: true });
const zip = path.join(dir, 'artifact.zip');
fetcher.downloadZip(opts.repo, artifact.id, zip);
const unzip = spawnSync('unzip', ['-o', '-q', zip, '-d', dir], { timeout: 120_000 });
if (unzip.status !== 0) throw new Error(`unzip failed for ${artifact.name}: ${String(unzip.stderr || unzip.error || '')}`);
fs.rmSync(zip, { force: true });
fs.writeFileSync(path.join(dir, '.complete'), '');
}
dirs.push(dir);
}
return dirs;
}
function gitOutput(args: string[]): string | null {
const result = spawnSync('git', args, { encoding: 'utf8', timeout: 5_000 });
return result.status === 0 ? result.stdout.trim() : null;
}
function repoSlug(): string {
const url = gitOutput(['remote', 'get-url', 'origin']) ?? '';
return url.match(/[:/]([^/:]+\/[^/]+?)(?:\.git)?$/)?.[1] ?? 'garrytan/gstack';
}
if (import.meta.main) {
const argv = process.argv.slice(2);
const dirFlag = argv.indexOf('--dir');
const dir = dirFlag !== -1 ? argv[dirFlag + 1] : getProjectEvalDir();
const flag = (name: string) => { const index = argv.indexOf(name); return index === -1 ? undefined : argv[index + 1]; };
const dirs = argv.flatMap((arg, index) => arg === '--dir' && argv[index + 1] ? [argv[index + 1]!] : []);
const asJson = argv.includes('--json');
const sinceFlag = argv.indexOf('--since-days');
const sinceDays = sinceFlag !== -1 ? Number(argv[sinceFlag + 1]) || 60 : 60;
const gate = argv.includes('--gate');
const backfill = argv.includes('--backfill');
const caseFilter = flag('--case');
const runsLimit = Number(flag('--runs')) || 10;
const sinceDays = Number(flag('--since-days')) || 60;
const repo = flag('--repo') ?? repoSlug();
const workflow = flag('--workflow') ?? 'evals-periodic.yml';
const branch = flag('--branch') ?? gitOutput(['rev-parse', '--abbrev-ref', 'HEAD']) ?? 'main';
const files = collectEvalFiles(dir, sinceDays);
const series = [...aggregate(files).values()]
.sort((a, b) => b.retriedPasses - a.retriedPasses || (b.fails / Math.max(1, b.runs)) - (a.fails / Math.max(1, a.runs)));
const ledger = readFreeLedger();
const records: TrialRecord[] = [];
const unattributed = new Set<string>();
const errors: string[] = [];
let historyError: string | null = null;
let weeklyRuns: string[] | undefined;
if (asJson) {
console.log(JSON.stringify({ dir, runsScanned: files.length, tests: series, freeLedger: ledger }, null, 2));
const manualReviews: string[] = [];
const importDir = (dir: string, run: { run_id: string; sha?: string; timestamp?: string } | undefined, legacyDays: number) => {
const trials = readTrialOutcomeDir(dir);
records.push(...trials.records);
errors.push(...trials.errors);
const legacy = backfillEvalFiles(collectEvalFiles(dir, legacyDays), run);
records.push(...legacy.records);
manualReviews.push(...legacy.manualReviews);
legacy.unattributed.forEach(name => unattributed.add(name));
};
if (dirs.length) {
for (const dir of dirs) importDir(dir, undefined, sinceDays);
} else {
console.log(`flake-rank: ${files.length} finalized run file(s) under ${dir}`);
const flaky = series.filter((s) => s.retriedPasses > 0 || s.fails > 0 || s.manualAccepted > 0);
if (flaky.length === 0) {
console.log(' no retried passes and no failures recorded — clean series');
} else {
console.log(' retries fails/runs manual avg-dur test');
for (const s of flaky.slice(0, 30)) {
console.log(` ${String(s.retriedPasses).padStart(7)} ${String(s.fails).padStart(5)}/${String(s.runs).padEnd(6)} `
+ `${String(s.manualAccepted).padStart(6)} ${Math.round(s.totalDurationMs / s.totalAttempts / 1000).toString().padStart(5)}s ${s.name}`);
try {
const runs = listWeeklyRuns({ repo, workflow, branches: [...new Set([branch, 'main'])], limit: runsLimit });
weeklyRuns = runs.map(run => run.createdAt);
const cacheDir = path.join(os.homedir(), '.gstack', 'eval-pass-rates-cache', repo.replace('/', '-'));
const match = backfill
? (name: string) => name.startsWith('trial-outcomes') || /^(paid-slice-\d+|gate-census-\d+)$/.test(name)
: (name: string) => name.startsWith('trial-outcomes');
for (const run of runs) {
const dirsForRun = downloadRunArtifacts({ repo, run, match, cacheDir, maxBytes: backfill ? 64 * 1024 * 1024 : undefined });
for (const dir of dirsForRun) importDir(dir, { run_id: `${run.id}`, sha: run.sha, timestamp: run.createdAt }, 3650);
}
} catch (error) {
historyError = error instanceof Error ? error.message : String(error);
}
}
const report = analyzePassRates(records, { weeklyRuns, unattributed: [...unattributed].sort(), errors, manualReviews });
const ledger = readFreeLedger();
if (asJson) {
console.log(JSON.stringify({ repo, workflow, branch, dirs, historyError, ...report, freeLedger: ledger }, null, 2));
} else {
if (historyError) console.log(`pass-rates: history unavailable (${historyError}); every label below is INCONCLUSIVE`);
console.log(formatPassRates(report, { caseFilter }));
if (ledger.length > 0) {
const byFile = new Map<string, number>();
for (const e of ledger) byFile.set(e.file, (byFile.get(e.file) ?? 0) + 1);
@@ -147,4 +698,5 @@ if (import.meta.main) {
}
}
}
if (gate && (historyError || report.alarms.length)) process.exit(1);
}
+306 -3
View File
@@ -13,6 +13,13 @@ import * as path from 'node:path';
import { spawnSync } from 'node:child_process';
import { aggregate, collectEvalFiles } from '../scripts/eval-flake-rank';
import { manualReviewFixture } from './helpers/manual-judge-review-fixture';
import {
analyzePassRates, attributeLegacyRecord, backfillEvalFiles, caseSeriesIdentities, downloadRunArtifacts, fisherOneSidedLower,
formatPassRates, holmRejections, listWeeklyRuns, quarantinePolicyProblems, quarantineRunsSince, readTrialOutcomeDir,
wilsonInterval, type HistoryFetcher, type PassRatePolicy, type QuarantineEntry, type Registry, type TrialRecord,
} from '../scripts/eval-flake-rank';
import { EVAL_POLICY } from './helpers/periodic-exclude-data';
import { TRIAL_OUTCOME_SCHEMA, formatTrialOutcomes } from './helpers/eval-store';
const entry = (name: string, passed: boolean, attempt: number) => ({
name, suite: 's', tier: 'e2e', passed, attempt, duration_ms: 1000, cost_usd: 0.1,
@@ -39,9 +46,10 @@ describe('eval-flake-rank aggregate', () => {
const display = spawnSync(process.execPath, [path.resolve(import.meta.dir, '../scripts/eval-flake-rank.ts'), '--dir', dir],
{ encoding: 'utf8', timeout: 10_000 });
expect(display.status, display.stderr).toBe(0);
expect(display.stdout).toContain('fails/runs manual');
expect(display.stdout).toContain('0/1');
expect(display.stdout).toContain(manual.name);
// pass-rates view: the prior automated pass is the one scored pre-policy
// trial; the manual acceptance is counted in its own column, never scored.
expect(display.stdout).toContain('pre-policy manual case');
expect(display.stdout).toMatch(new RegExp(`1/1 \\[[^\\]]+\\]\\s+1 ${manual.name.replace(/[.*+?^${}()|[\]\\/]/g, '\\$&')}`));
fs.writeFileSync(path.join(dir, 'invalid-retry.json'), run([
{ ...ordinary, attempt: 1 }, { ...manual, attempt: 2 },
]));
@@ -97,3 +105,298 @@ describe('eval-flake-rank aggregate', () => {
fs.rmSync(dir, { recursive: true, force: true });
});
});
// --- pass-rates ---
const registry: Registry = {
kinds: { 'rule-a': 'rule', 'beh-b': 'behavior', 'gate-c': 'rule', 'mar-d': 'rule', 'judge one': 'judge',
...Object.fromEntries(Array.from({ length: 18 }, (_, i) => [`filler-${i}`, 'rule'])) },
tiers: { 'rule-a': 'periodic', 'beh-b': 'periodic', 'gate-c': 'gate', 'mar-d': 'marathon',
...Object.fromEntries(Array.from({ length: 18 }, (_, i) => [`filler-${i}`, i < 9 ? 'gate' : 'periodic'])) },
touchfiles: { 'rule-a': ['test/skill-e2e-a.test.ts', 'a/**'], 'beh-b': ['test/skill-e2e-b.test.ts', 'b/**'],
'gate-c': ['test/skill-e2e-shared.test.ts'], 'mar-d': ['test/skill-e2e-shared.test.ts'] },
judgeTouchfiles: { 'judge one': ['j/SKILL.md'] },
globals: ['harness/**'],
testNames: { 'gate-c': '/gate c labeled' },
};
let clock = 0;
function trial(id: string, outcome: 'passed' | 'failed' | 'skipped', extra: Partial<TrialRecord> = {}): TrialRecord {
clock += 1;
return {
schema: TRIAL_OUTCOME_SCHEMA, case: id, file: 'test/x.test.ts', tier: registry.tiers[id] ?? 'judge',
kind: registry.kinds[id]!, trial: 1, panel: { n: 1, k: 1 }, attempt: 1, outcome,
...(outcome === 'failed' ? { failure_class: 'assertion' as const } : {}),
duration_ms: 1, cost_usd: 0, model: 'model-x', cli_version: '2.1.284', policy_version: 1, quarantined: false,
execution: 'executed', source: 'shard', run_id: `run-${clock}`, recorded_at: new Date(Date.UTC(2026, 9, 1) + clock * 60_000).toISOString(),
series_identity: 'id-1', ...extra,
};
}
const many = (id: string, passes: number, fails: number, extra: Partial<TrialRecord> = {}) =>
[...Array.from({ length: passes }, () => trial(id, 'passed', extra)), ...Array.from({ length: fails }, () => trial(id, 'failed', extra))];
const analyze = (records: TrialRecord[], quarantine: Record<string, QuarantineEntry> = {}, extra = {}) =>
analyzePassRates(records, { registry, quarantine, now: Date.UTC(2026, 9, 2), ...extra });
const qEntry = (overrides: Partial<QuarantineEntry> = {}): QuarantineEntry => ({
reason: 'Detector graded the posture wording; 7 of 10 fresh trials failed only the regex, transcripts attached.',
failureClass: 'detector', tracking: 'TODOS.md "x"', owner: 'garrytan', enteredAt: '2026-09-29',
exit: '>= 97% over >= 10 trials on the current identity', ...overrides,
});
describe('pass-rates statistics', () => {
test('Wilson bounds match the documented policy arithmetic', () => {
expect(wilsonInterval(10, 10).lo).toBeCloseTo(0.7225, 4);
expect(wilsonInterval(6, 6).lo).toBeCloseTo(0.6097, 4);
expect(wilsonInterval(125, 125).lo).toBeCloseTo(0.9702, 4);
expect(wilsonInterval(10, 10).hi).toBe(1);
expect(wilsonInterval(0, 0)).toEqual({ lo: 0, hi: 1 });
const mid = wilsonInterval(7, 10);
expect(mid.lo).toBeGreaterThan(0.39); expect(mid.hi).toBeLessThan(0.9);
});
test('one-sided Fisher exact matches a known table and is one-sided', () => {
expect(fisherOneSidedLower(4, 6, 6, 6)).toBeCloseTo(0.227272727, 8);
expect(fisherOneSidedLower(6, 6, 4, 6)).toBe(1);
expect(fisherOneSidedLower(0, 10, 10, 10)).toBeLessThan(1e-4);
});
test('Holm rejects step-down and stops at the first non-rejection', () => {
expect([...holmRejections([0.001, 0.02, 0.04], 0.05)].sort()).toEqual([0, 1, 2]);
expect([...holmRejections([0.001, 0.03, 0.04], 0.05)]).toEqual([0]);
expect([...holmRejections([0.03, 0.04], 0.05)]).toEqual([]);
expect([...holmRejections([], 0.05)]).toEqual([]);
});
});
describe('pass-rates labels', () => {
test('thin history is INCONCLUSIVE, and after this PR every series starts there', () => {
const report = analyze(many('rule-a', 9, 0));
expect(report.cases[0]).toMatchObject({ case: 'rule-a', label: 'INCONCLUSIVE' });
expect(formatPassRates(analyze([]))).toContain('every series starts INCONCLUSIVE');
});
test('PASSING, FLAKY and FAILING come from the interval against the entry rate', () => {
expect(analyze(many('rule-a', 12, 0)).cases[0]!.label).toBe('PASSING');
expect(analyze(many('beh-b', 10, 1)).cases[0]!.label).toBe('FLAKY');
expect(analyze(many('beh-b', 2, 10)).cases[0]!.label).toBe('FAILING');
});
test('BROKEN: the latest run is 0/n after a prior interval at or above the entry rate', () => {
const prior = many('beh-b', 80, 0, { run_id: 'old' });
const latest = [1, 2, 3].map(n => trial('beh-b', 'failed', { run_id: 'new', trial: n, panel: { n: 3, k: 2 } }));
expect(analyze([...prior, ...latest]).cases[0]!.label).toBe('BROKEN');
});
test('skipped trials carry no verdict; infra failures count as failed trials', () => {
const stats = analyze([...many('rule-a', 10, 0), trial('rule-a', 'skipped'),
trial('rule-a', 'failed', { failure_class: 'infra' })]).cases[0]!.current!;
expect(stats).toMatchObject({ passes: 10, trials: 11, infra: 1 });
});
test('a new identity, model or CLI starts a new series; earlier series stay visible', () => {
const report = analyze([...many('rule-a', 10, 0), ...many('rule-a', 3, 0, { series_identity: 'id-2' }),
...many('rule-a', 2, 0, { series_identity: 'id-2', cli_version: '2.1.285' })]);
const c = report.cases[0]!;
expect(c.series).toHaveLength(3);
expect(c.current).toMatchObject({ identity: 'id-2', cli: '2.1.285', trials: 2 });
expect(c.previous).toMatchObject({ identity: 'id-2', cli: '2.1.284', trials: 3 });
expect(c.label).toBe('INCONCLUSIVE');
});
});
describe('pass-rates alarms count post-policy trials of the current series only', () => {
test('backfilled pre-policy failures are displayed but never alarm', () => {
const report = analyze(many('rule-a', 2, 20, { policy_version: 0, source: 'backfill' }));
expect(report.alarms).toEqual([]);
expect(report.cases[0]!.prePolicy).toMatchObject({ passes: 2, trials: 22 });
expect(report.cases[0]!.label).toBe('INCONCLUSIVE');
});
test('drift proposes quarantine for a blocking case below the entry rule; a rule case is flagged as behaving like behavior', () => {
const kinds = analyze([...many('rule-a', 8, 2), ...many('beh-b', 8, 2), ...many('mar-d', 0, 10)]).alarms.map(a => `${a.kind}:${a.case}`);
expect([...kinds].sort()).toEqual(['drift:beh-b', 'drift:rule-a', 'rule-as-behavior:mar-d', 'rule-as-behavior:rule-a']);
expect(analyze(many('rule-a', 19, 1)).alarms).toEqual([]);
});
test('the Fisher regression alarm needs the minimum trials on both sides', () => {
const old = many('gate-c', 6, 0, { series_identity: 'old' });
const fresh = many('gate-c', 0, 6, { series_identity: 'new' });
expect(analyze([...old, ...fresh]).alarms.map(a => a.kind)).toContain('regression');
expect(analyze([...old, ...fresh.slice(0, 5)]).alarms.map(a => a.kind)).not.toContain('regression');
});
test('quarantine exit, expiry and cap', () => {
const exit = analyze(many('beh-b', 10, 0), { 'beh-b': qEntry() }).alarms.map(a => a.kind);
expect(exit).toContain('quarantine-exit');
expect(exit).not.toContain('drift');
const weekly = Array.from({ length: 8 }, (_, i) => new Date(Date.UTC(2026, 8, 30) + i * 7 * 86_400_000).toISOString());
expect(analyze([], { 'beh-b': qEntry() }, { weeklyRuns: weekly }).alarms.map(a => a.kind)).toContain('quarantine-expired');
expect(analyze([], { 'beh-b': qEntry() }, { weeklyRuns: weekly.slice(0, 7) }).alarms.map(a => a.kind)).not.toContain('quarantine-expired');
expect(quarantineRunsSince('2026-09-01', undefined, Date.UTC(2026, 9, 27))).toBe(8);
expect(quarantineRunsSince('not a date', undefined, 0)).toBe(Number.POSITIVE_INFINITY);
});
});
describe('quarantine policy', () => {
const policy: PassRatePolicy = EVAL_POLICY;
const now = Date.UTC(2026, 9, 2);
test('a valid entry has no problems', () => {
expect(quarantinePolicyProblems({ 'beh-b': qEntry() }, registry, policy, now)).toEqual([]);
});
test('a product defect, a missing diagnosis or field, a bad date or a non-blocking case is rejected', () => {
const problems = (quarantine: Record<string, QuarantineEntry>) => quarantinePolicyProblems(quarantine, registry, policy, now).map(p => p.message);
expect(problems({ 'beh-b': qEntry({ failureClass: 'product' as QuarantineEntry['failureClass'] }) }).join()).toContain('never quarantined');
expect(problems({ 'beh-b': qEntry({ reason: 'flaky' }) }).join()).toContain('written diagnosis');
expect(problems({ 'beh-b': qEntry({ owner: ' ' }) }).join()).toContain('missing owner');
expect(problems({ 'beh-b': qEntry({ enteredAt: '09/29/2026' }) }).join()).toContain('YYYY-MM-DD');
expect(problems({ 'beh-b': qEntry({ enteredAt: '2027-01-01' }) }).join()).toContain('future');
expect(problems({ 'mar-d': qEntry() }).join()).toContain('not blocking');
expect(problems({ 'judge one': qEntry() }).join()).toContain('no registered E2E case');
expect(problems({ ghost: qEntry() }).join()).toContain('no registered E2E case');
});
test('at most 10% of a tier may be quarantined', () => {
// 11 periodic cases in the fixture registry: the cap is 1.
expect(quarantinePolicyProblems({ 'beh-b': qEntry() }, registry, policy, now)).toEqual([]);
const over = quarantinePolicyProblems({ 'beh-b': qEntry(), 'rule-a': qEntry() }, registry, policy, now);
expect(over.map(p => p.kind)).toEqual(['quarantine-cap']);
expect(over[0]!.message).toContain('2 quarantined periodic cases exceed the 10% cap (1 of 11)');
});
});
describe('pass-rates inputs', () => {
test('trial-outcomes JSONL is schema-validated; invalid lines are reported, never guessed', () => {
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'passrates-'));
const valid = trial('rule-a', 'passed');
fs.mkdirSync(path.join(dir, 'nested'));
fs.writeFileSync(path.join(dir, 'nested', 'trial-outcomes.jsonl'), formatTrialOutcomes([valid]) + '{"schema":"other"}\nnot json\n');
fs.writeFileSync(path.join(dir, 'unrelated.jsonl'), formatTrialOutcomes([valid]));
const read = readTrialOutcomeDir(dir);
expect(read.records).toHaveLength(1);
expect(read.records[0]).toMatchObject({ case: 'rule-a', series_identity: 'id-1' });
expect(read.errors).toHaveLength(2);
fs.rmSync(dir, { recursive: true, force: true });
});
test('legacy records attribute by shard suffix, id, label or single-owner file, else stay unattributed', () => {
expect(attributeLegacyRecord('/anything', 'skill-e2e-b--beh-b', registry)).toBe('beh-b');
expect(attributeLegacyRecord('rule-a', 'skill-e2e-zzz', registry)).toBe('rule-a');
expect(attributeLegacyRecord('/gate c labeled', undefined, registry)).toBe('gate-c');
expect(attributeLegacyRecord('/a display name', 'skill-e2e-a', registry)).toBe('rule-a');
expect(attributeLegacyRecord('/shared display', 'skill-e2e-shared', registry)).toBeNull();
});
test('backfill keeps only first attempts, defaults a missing attempt to 1, and labels records pre-policy', () => {
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'passrates-backfill-'));
fs.writeFileSync(path.join(dir, 'run.json'), run([
{ ...entry_('rule-a', false, 1), exit_reason: 'timeout' }, entry_('rule-a', true, 2),
{ name: 'beh-b', suite: 's', tier: 'e2e', passed: true, duration_ms: 1, cost_usd: 0 },
entry_('/unknown display', true, 1),
], { shard: 'skill-e2e-zzz' }));
const { records, unattributed } = backfillEvalFiles(collectEvalFiles(dir), { run_id: '42', sha: 'abc' }, registry);
expect(records.map(r => [r.case, r.outcome, r.failure_class, r.policy_version, r.source, r.run_id]))
.toEqual([['rule-a', 'failed', 'timeout', 0, 'backfill', '42'], ['beh-b', 'passed', undefined, 0, 'backfill', '42']]);
expect(formatTrialOutcomes(records)).toContain(TRIAL_OUTCOME_SCHEMA);
expect(unattributed).toEqual(['/unknown display']);
fs.rmSync(dir, { recursive: true, force: true });
});
test('series identity follows the case\'s own touchfiles, not GLOBAL_TOUCHFILES', () => {
const root = fs.mkdtempSync(path.join(os.tmpdir(), 'passrates-series-'));
const git = (...args: string[]) => spawnSync('git', args, { cwd: root, encoding: 'utf8', timeout: 10_000 });
for (const [file, body] of [['a/x.ts', '1'], ['b/y.ts', '1'], ['harness/run.ts', '1'], ['test/skill-e2e-a.test.ts', '1']] as const) {
fs.mkdirSync(path.join(root, path.dirname(file)), { recursive: true });
fs.writeFileSync(path.join(root, file), body);
}
const snapshot = () => { expect(git('add', '-A').status).toBe(0); return caseSeriesIdentities(['rule-a', 'beh-b'], root, registry); };
expect(git('init', '-q').status).toBe(0);
const first = snapshot();
expect(first['rule-a']).not.toBe(first['beh-b']);
fs.writeFileSync(path.join(root, 'harness/run.ts'), '2');
expect(snapshot()).toEqual(first);
fs.writeFileSync(path.join(root, 'a/x.ts'), '2');
const next = snapshot();
expect(next['rule-a']).not.toBe(first['rule-a']);
expect(next['beh-b']).toBe(first['beh-b']);
fs.rmSync(root, { recursive: true, force: true });
});
});
describe('pass-rates history fetch (injected, no network)', () => {
function storedZip(files: Record<string, string>): Buffer {
const locals: Buffer[] = [], centrals: Buffer[] = [];
let offset = 0;
for (const [name, text] of Object.entries(files)) {
const data = Buffer.from(text), fileName = Buffer.from(name), crc = Bun.hash.crc32(data) >>> 0;
const local = Buffer.alloc(30); local.writeUInt32LE(0x04034b50, 0); local.writeUInt16LE(20, 4);
local.writeUInt32LE(crc, 14); local.writeUInt32LE(data.length, 18); local.writeUInt32LE(data.length, 22); local.writeUInt16LE(fileName.length, 26);
const central = Buffer.alloc(46); central.writeUInt32LE(0x02014b50, 0); central.writeUInt16LE(20, 4); central.writeUInt16LE(20, 6);
central.writeUInt32LE(crc, 16); central.writeUInt32LE(data.length, 20); central.writeUInt32LE(data.length, 24);
central.writeUInt16LE(fileName.length, 28); central.writeUInt32LE(offset, 42);
locals.push(local, fileName, data); centrals.push(central, fileName);
offset += 30 + fileName.length + data.length;
}
const size = centrals.reduce((sum, b) => sum + b.length, 0);
const end = Buffer.alloc(22); end.writeUInt32LE(0x06054b50, 0); end.writeUInt16LE(Object.keys(files).length, 8);
end.writeUInt16LE(Object.keys(files).length, 10); end.writeUInt32LE(size, 12); end.writeUInt32LE(offset, 16);
return Buffer.concat([...locals, ...centrals, end]);
}
test('lists runs per branch, deduplicated and newest first', () => {
const fetcher: HistoryFetcher = {
listRuns: (_repo, _workflow, branch) => branch === 'main'
? [{ id: 1, attempt: 1, sha: 'a', branch, createdAt: '2026-09-01T00:00:00Z' }, { id: 3, attempt: 1, sha: 'c', branch, createdAt: '2026-09-15T00:00:00Z' }]
: [{ id: 3, attempt: 1, sha: 'c', branch, createdAt: '2026-09-15T00:00:00Z' }, { id: 2, attempt: 2, sha: 'b', branch, createdAt: '2026-09-08T00:00:00Z' }],
listArtifacts: () => [], downloadZip: () => { throw new Error('unused'); },
};
expect(listWeeklyRuns({ repo: 'o/r', workflow: 'evals-periodic.yml', branches: ['feature', 'main'], limit: 10, fetcher }).map(r => r.id)).toEqual([3, 2, 1]);
});
test('downloads only matching, bounded artifacts once, and caches them', () => {
const cacheDir = fs.mkdtempSync(path.join(os.tmpdir(), 'passrates-cache-'));
const downloads: number[] = [];
const fetcher: HistoryFetcher = {
listRuns: () => [],
listArtifacts: () => [{ id: 10, name: 'trial-outcomes-gate', size: 100 }, { id: 11, name: 'paid-slice-1', size: 100 },
{ id: 12, name: 'trial-outcomes-huge', size: 10 ** 9 }, { id: 13, name: 'trial-outcomes/../escape', size: 1 }],
downloadZip: (_repo, id, destination) => { downloads.push(id); fs.writeFileSync(destination, storedZip({ 'trial-outcomes.jsonl': formatTrialOutcomes([trial('rule-a', 'passed')]) })); },
};
const options = { repo: 'o/r', run: { id: 7, attempt: 1, sha: 's', branch: 'main', createdAt: '' }, cacheDir, fetcher,
match: (name: string) => name.startsWith('trial-outcomes') };
const dirs = downloadRunArtifacts(options);
expect(downloads).toEqual([10]);
expect(dirs).toHaveLength(1);
expect(readTrialOutcomeDir(dirs[0]!).records.map(r => r.case)).toEqual(['rule-a']);
expect(downloadRunArtifacts(options)).toEqual(dirs);
expect(downloads).toEqual([10]);
fs.rmSync(cacheDir, { recursive: true, force: true });
});
});
describe('pass-rates CLI', () => {
const cli = (args: string[]) => spawnSync(process.execPath, [path.resolve(import.meta.dir, '../scripts/eval-flake-rank.ts'), ...args],
{ encoding: 'utf8', timeout: 20_000 });
test('--dir prints per-case pass rates; --gate fails only on ACTION REQUIRED', () => {
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'passrates-cli-'));
const id = 'plan-ceo-review-format-mode';
const records = Array.from({ length: 12 }, (_, i) => ({ ...trial(id, i < 11 ? 'passed' : 'failed'), kind: 'behavior' as const, tier: 'periodic' }));
fs.writeFileSync(path.join(dir, 'trial-outcomes.jsonl'), formatTrialOutcomes(records));
const shown = cli(['--dir', dir, '--case', id]);
expect(shown.status, shown.stderr).toBe(0);
expect(shown.stdout).toContain(`11/12 [`);
expect(shown.stdout).toMatch(new RegExp(`FLAKY\\s+behavior\\s+periodic.*${id}`));
expect(shown.stdout).toContain('ACTION REQUIRED');
expect(shown.stdout).toContain(`[drift] ${id} passes 11/12`);
expect(cli(['--dir', dir, '--gate']).status).toBe(1);
fs.writeFileSync(path.join(dir, 'trial-outcomes.jsonl'), formatTrialOutcomes(records.slice(0, 11)));
const clean = cli(['--dir', dir, '--gate', '--json']);
expect(clean.status, clean.stdout).toBe(0);
expect(JSON.parse(clean.stdout).cases[0]).toMatchObject({ case: id, label: 'PASSING', current: { passes: 11, trials: 11 } });
fs.rmSync(dir, { recursive: true, force: true });
});
});
function entry_(name: string, passed: boolean, attempt: number) {
return { name, suite: 's', tier: 'e2e', passed, attempt, duration_ms: 1000, cost_usd: 0.1 };
}
+9 -5
View File
@@ -106,14 +106,18 @@ export const EVAL_POLICY = {
* failures (a product defect is never quarantined), and unchanged case
* touchfiles in the change that adds it. Pinned by
* test/periodic-exclude-policy.test.ts.
* reason - the written diagnosis
* tracking - issue or TODOS pointer
* owner - who removes it
* enteredAt - ISO date the entry landed (expiry counts weekly runs from here)
* exit - the measurable exit condition
* reason - the written diagnosis, with the pass-rate evidence
* failureClass - what the diagnosis found; a product defect has no class here
* tracking - issue or TODOS pointer
* owner - who removes it
* enteredAt - YYYY-MM-DD the entry landed (expiry counts weekly runs from here)
* exit - the measurable exit condition
* At most EVAL_POLICY.quarantine.capFraction of a tier's cases (gate and
* periodic are the blocking tiers) may be quarantined at once.
*/
export const CASE_QUARANTINE: Record<string, {
reason: string;
failureClass: 'detector' | 'harness' | 'model-latency';
tracking: string;
owner: string;
enteredAt: string;
+30 -2
View File
@@ -9,10 +9,11 @@ import { describe, expect, test } from 'bun:test';
import * as fs from 'node:fs';
import * as path from 'node:path';
import { CASE_CI_EXCLUDE, PERIODIC_CI_EXCLUDE } from './helpers/periodic-exclude-data';
import { CASE_CI_EXCLUDE, CASE_QUARANTINE, EVAL_POLICY, PERIODIC_CI_EXCLUDE } from './helpers/periodic-exclude-data';
import { E2E_TOUCHFILES } from './helpers/touchfiles';
import { quarantinePolicyProblems } from '../scripts/eval-flake-rank';
import { isPaidTestFile } from './helpers/paid-test-set';
import { buildRunManifest, CASE_SHARDED_FILES, expandCaseShards, partitionCaseExclusions, selectPaidTestFiles, shardCaseId, shardFile } from '../scripts/test-paid-shards';
import { buildRunManifest, CASE_SHARDED_FILES, expandCaseShards, fileCaseRegistration, partitionCaseExclusions, selectPaidTestFiles, shardCaseId, shardFile } from '../scripts/test-paid-shards';
const ROOT = path.resolve(__dirname, '..');
@@ -72,3 +73,30 @@ describe('periodic exclude policy', () => {
}
});
});
describe('eval verdict policy (pre-registered)', () => {
test('EVAL_POLICY carries exactly the approved constants; a change needs re-approval and a version bump', () => {
expect(EVAL_POLICY).toEqual({
version: 1,
panel: { n: 3, k: 2 },
quarantine: { entry: { rate: 0.95, minTrials: 10 }, exit: { rate: 0.97, minTrials: 10 }, capFraction: 0.10, expiryWeeklyRuns: 8 },
judge: { samples: 3 },
drift: { fisherAlpha: 0.05, fisherMinPerSide: 6 },
infraRedispatch: 1,
});
});
test('every CASE_QUARANTINE entry is a diagnosed, dated, non-product blocking case within the tier cap', () => {
expect(quarantinePolicyProblems(CASE_QUARANTINE).map(problem => problem.message)).toEqual([]);
});
test('a quarantined case runs as isolated trial shards: its files register it literally', () => {
for (const id of Object.keys(CASE_QUARANTINE)) {
const files = (E2E_TOUCHFILES[id] ?? []).filter(file => /^test\/[^/]+\.test\.ts$/.test(file) && isPaidTestFile(file));
expect(files.length, `${id}: no paid file registers it`).toBeGreaterThan(0);
for (const file of files) {
expect(fileCaseRegistration(file, fs.readFileSync(path.join(ROOT, file), 'utf8')).known, `${id}: ${file} registration must be statically known`).toBe(true);
}
}
});
});