diff --git a/scripts/eval-flake-rank.ts b/scripts/eval-flake-rank.ts index 3e6028a10..ca46a3eb8 100644 --- a/scripts/eval-flake-rank.ts +++ b/scripts/eval-flake-rank.ts @@ -1,29 +1,55 @@ #!/usr/bin/env bun /** - * eval-flake-rank — the flake-telemetry dial (WS1). + * eval-pass-rates (alias: eval-flake-rank) — per-case trial pass rates. * - * Aggregates per-test series across every FINALIZED eval-store run on this - * machine (default: ~/.gstack/projects//evals/, shard dirs included) - * plus the free suite's flake ledger, and ranks tests by flake signal: - * retried passes first (a test that needs attempt 2 to go green is the - * definition of a flake), then failure rate. + * Reads trial records (one JSONL line per trial: case, kind, trial, outcome, + * exit_reason, duration, cost, model, CLI version, series identity, run id, + * sha, policy_version) from the last N completed `evals-periodic.yml` runs on + * the current branch and `main` (downloading only each run's small + * `trial-outcomes` artifact through `gh`), plus any local eval dirs, and + * prints per-case per-trial pass rates with 95% Wilson intervals. * - * This is the readable dial behind two policies: - * - a flaky pass never blocks a merge, but it is recorded and RANKED here; - * - the required-check promotion (WS16) needs weeks of clean flake-rank, - * not vibes. + * A series is one case under one input identity: the case's own touchfiles + * minus GLOBAL_TOUCHFILES (`caseSeriesIdentities`), grouped by model and CLI + * version, per policy_version. A new identity starts a new series; earlier + * series stay visible. Only post-policy trials of the current series feed the + * labels and alarms. Legacy eval-store records (`--backfill`, `--dir`) are + * imported as pre-policy trials (first attempt only; a missing attempt means + * 1) and are display-only. + * + * Labels: INCONCLUSIVE (below the entry rule's minimum trials), BROKEN (latest run 0/n + * after a prior interval at or above the entry rate), FLAKY (failures and an + * interval straddling the entry rate), FAILING (interval below the entry + * rate), PASSING (otherwise). + * + * The weekly gate (`--gate`) exits non-zero with ACTION REQUIRED when a + * non-quarantined case meets the quarantine entry rule, a rule case behaves + * like a behavior case, a blocking case's current-identity rate is + * significantly below its previous identity (one-sided Fisher exact, + * Holm-controlled across cases), or a CASE_QUARANTINE entry has met its exit + * rule, expired, or pushed its tier over the cap. History that cannot be + * fetched fails the gate closed. * * Usage: - * bun run eval:flake-rank # project eval dir - * bun run eval:flake-rank --dir # e.g. downloaded CI artifacts - * bun run eval:flake-rank --json # machine-readable + * bun run eval:pass-rates # last 10 weekly runs, this branch + main + * bun run eval:pass-rates --case --runs 20 + * bun run eval:pass-rates --dir # local eval dirs / downloaded artifacts (repeatable) + * bun run eval:pass-rates --backfill # also import legacy slice artifacts, labeled pre-policy + * bun run eval:pass-rates --json | --gate */ import * as fs from 'node:fs'; +import * as os from 'node:os'; import * as path from 'node:path'; -import { getProjectEvalDir, isPartialEval, isFinalizedEvalResultFile, type EvalResult } from '../test/helpers/eval-store'; -import { evalEntryOutcome } from '../test/helpers/eval-store'; +import { spawnSync } from 'node:child_process'; +import { createHash } from 'node:crypto'; +import { isPartialEval, isFinalizedEvalResultFile, evalEntryOutcome, failureClassOf, parseTrialOutcomes, sanitizeTrialError, + TRIAL_OUTCOME_SCHEMA, type EvalCaseKind, type EvalResult, type TrialOutcomeRecord } from '../test/helpers/eval-store'; import { flakeLedgerPath, type FlakeLedgerEntry } from './test-free-shards'; +import { E2E_KINDS, E2E_TIERS, E2E_TOUCHFILES, GLOBAL_TOUCHFILES, LLM_JUDGE_TOUCHFILES } from '../test/helpers/touchfiles-data'; +import { CASE_QUARANTINE, EVAL_POLICY } from '../test/helpers/periodic-exclude-data'; +import { matchGlob } from '../test/helpers/test-selection'; +import { CASE_TEST_NAMES } from './test-paid-shards'; interface TestSeries { name: string; @@ -111,33 +137,558 @@ function readFreeLedger(): FlakeLedgerEntry[] { return out; } +// --- Trial records --- + +/** + * A trial record as pass-rates reads it: eval-store's trial-outcomes schema + * plus the series identity the report job stamps (caseSeriesIdentities). + * policy_version 0 marks a pre-policy (backfilled) record. + */ +export type TrialRecord = TrialOutcomeRecord & { series_identity?: string }; + +/** Per-file cap for downloaded artifacts: pass-rates parses data only, never executes it. */ +export const TRIAL_OUTCOMES_MAX_BYTES = 8 * 1024 * 1024; + +/** Every `trial-outcomes*.jsonl` file under a directory, size-capped, schema-validated by eval-store. */ +export function readTrialOutcomeDir(dir: string): { records: TrialRecord[]; errors: string[] } { + const records: TrialRecord[] = []; + const errors: string[] = []; + if (!fs.existsSync(dir)) return { records, errors }; + for (const name of fs.readdirSync(dir, { recursive: true }) as string[]) { + if (!/(^|\/)trial-outcomes[^/]*\.jsonl$/.test(name)) continue; + const full = path.join(dir, name); + const parsed = parseTrialOutcomes(fs.readFileSync(full, 'utf8'), { maxBytes: TRIAL_OUTCOMES_MAX_BYTES }); + records.push(...parsed.records.map(record => ({ + ...record, series_identity: typeof (record as TrialRecord).series_identity === 'string' + ? (record as TrialRecord).series_identity!.slice(0, 64) : undefined }))); + errors.push(...parsed.errors.map(error => `${full}: ${error}`)); + } + return { records, errors }; +} + +// --- Registry attribution and series identity --- + +export interface Registry { + kinds: Record; + tiers: Record; + touchfiles: Record; + judgeTouchfiles: Record; + globals: readonly string[]; + testNames: Record; +} + +export const LIVE_REGISTRY: Registry = { + kinds: E2E_KINDS, tiers: E2E_TIERS, touchfiles: E2E_TOUCHFILES, judgeTouchfiles: LLM_JUDGE_TOUCHFILES, + globals: GLOBAL_TOUCHFILES, testNames: CASE_TEST_NAMES, +}; + +/** A case's tier: its E2E_TIERS value, or 'judge' for an LLM-judge entry. */ +export function caseTier(id: string, registry: Registry = LIVE_REGISTRY): string { + return registry.tiers[id] ?? (id in registry.judgeTouchfiles ? 'judge' : 'unknown'); +} + +/** + * Attribute a legacy eval-store record to a registry id: the case-shard slug + * suffix (`--`), the recorded name, a CASE_TEST_NAMES label, or the + * only id its shard file registers. Anything else is unattributed (null). + */ +export function attributeLegacyRecord(name: string, shard: string | undefined, registry: Registry = LIVE_REGISTRY): string | null { + const known = (id: string) => id in registry.kinds; + const [slugFile, slugCase] = (shard ?? '').split('--'); + if (slugCase && known(slugCase)) return slugCase; + if (known(name)) return name; + const labeled = Object.entries(registry.testNames).find(([, label]) => label === name)?.[0]; + if (labeled && known(labeled)) return labeled; + if (slugFile) { + const file = `test/${slugFile}.test.ts`; + const owners = Object.keys(registry.touchfiles).filter(id => registry.touchfiles[id]!.includes(file)); + if (owners.length === 1 && known(owners[0]!)) return owners[0]!; + } + return null; +} + +/** + * Series identity per case: a hash of the git blob ids of the files matching + * the case's own touchfiles, excluding GLOBAL_TOUCHFILES (harness edits are + * markers, not new series). The report job stamps this on every trial record. + */ +export function caseSeriesIdentities(ids: string[], root: string, registry: Registry = LIVE_REGISTRY): Record { + const listed = spawnSync('git', ['ls-files', '-s'], { cwd: root, encoding: 'utf8', timeout: 20_000, maxBuffer: 64 * 1024 * 1024 }); + if (listed.status !== 0) throw new Error(`git ls-files failed: ${listed.stderr}`); + const blobs = listed.stdout.split('\n').filter(Boolean).map(line => { + const [meta, file] = line.split('\t'); + return { file: file!, blob: meta!.split(' ')[1]! }; + }).filter(entry => !registry.globals.some(pattern => matchGlob(entry.file, pattern))); + return Object.fromEntries(ids.map(id => { + const patterns = registry.touchfiles[id] ?? registry.judgeTouchfiles[id] ?? []; + const lines = blobs.filter(entry => patterns.some(pattern => matchGlob(entry.file, pattern))) + .map(entry => `${entry.file} ${entry.blob}`).sort(); + return [id, createHash('sha256').update(`${id}\n${lines.join('\n')}`).digest('hex').slice(0, 16)]; + })); +} + +/** + * Import legacy eval-store result files as pre-policy trials (policy_version + * 0, source 'backfill'): first attempt only (a missing attempt means 1), + * attributed by registry id, never guessed. A manual-review acceptance carries + * no automated verdict: it is counted and shown, never scored. Without a CI + * run, each local result file is its own run. + */ +export function backfillEvalFiles(files: string[], run?: { run_id: string; sha?: string; timestamp?: string }, + registry: Registry = LIVE_REGISTRY): { records: TrialRecord[]; unattributed: string[]; manualReviews: string[] } { + const records: TrialRecord[] = []; + const unattributed = new Set(); + const manualReviews: string[] = []; + for (const file of files) { + let result: EvalResult & { shard?: string; claude_cli_version?: string }; + try { result = JSON.parse(fs.readFileSync(file, 'utf8')); } catch { continue; } + if (isPartialEval(result, file) || !Array.isArray(result.tests)) continue; + const seen = new Set(); + for (const entry of result.tests) { + if ((entry.attempt ?? 1) !== 1 || seen.has(entry.name)) continue; + seen.add(entry.name); + const id = attributeLegacyRecord(entry.name, result.shard, registry); + if (!id) { unattributed.add(entry.name); continue; } + const outcome = evalEntryOutcome(entry); + if (outcome === 'manual-review') { manualReviews.push(id); continue; } + records.push({ + schema: TRIAL_OUTCOME_SCHEMA, case: id, + file: result.shard ? `test/${result.shard.split('--')[0]}.test.ts` : 'unknown', + tier: caseTier(id, registry), kind: registry.kinds[id]!, trial: 1, panel: { n: 1, k: 1 }, attempt: 1, + outcome, ...(outcome === 'failed' ? { failure_class: failureClassOf(entry) } : {}), + exit_reason: entry.exit_reason, error: sanitizeTrialError(entry.error), + duration_ms: Math.max(0, entry.duration_ms || 0), cost_usd: Math.max(0, entry.cost_usd || 0), + model: entry.model, cli_version: result.claude_cli_version, policy_version: 0, quarantined: false, + execution: entry.execution === 'reused' ? 'reused' : 'executed', source: 'backfill', + run_id: run?.run_id ?? `local:${file}`, sha: run?.sha ?? result.git_sha, recorded_at: run?.timestamp ?? result.timestamp, + }); + } + } + return { records, unattributed: [...unattributed].sort(), manualReviews }; +} + +// --- Statistics --- + +/** 95% Wilson score interval for k successes in n trials. */ +export function wilsonInterval(k: number, n: number, z = 1.96): { lo: number; hi: number } { + if (n <= 0) return { lo: 0, hi: 1 }; + const p = k / n, z2 = z * z, denom = 1 + z2 / n; + const center = (p + z2 / (2 * n)) / denom; + const half = (z * Math.sqrt(p * (1 - p) / n + z2 / (4 * n * n))) / denom; + return { lo: Math.max(0, center - half), hi: k === n ? 1 : Math.min(1, center + half) }; +} + +function logChoose(n: number, k: number): number { + let sum = 0; + for (let i = 1; i <= k; i++) sum += Math.log(n - k + i) - Math.log(i); + return sum; +} + +/** + * One-sided Fisher exact p-value that the CURRENT pass rate is below the + * PREVIOUS one: P(X <= curPass) under the hypergeometric null with the + * observed margins. + */ +export function fisherOneSidedLower(curPass: number, curN: number, prevPass: number, prevN: number): number { + const passes = curPass + prevPass, total = curN + prevN; + const denom = logChoose(total, passes); + let p = 0; + for (let x = Math.max(0, passes - prevN); x <= curPass; x++) p += Math.exp(logChoose(curN, x) + logChoose(prevN, passes - x) - denom); + return Math.min(1, p); +} + +/** Holm step-down: the indices whose p-values are rejected at family-wise alpha. */ +export function holmRejections(pValues: number[], alpha: number): Set { + const order = pValues.map((p, index) => ({ p, index })).sort((a, b) => a.p - b.p); + const rejected = new Set(); + for (let rank = 0; rank < order.length; rank++) { + if (order[rank]!.p > alpha / (order.length - rank)) break; + rejected.add(order[rank]!.index); + } + return rejected; +} + +// --- Analysis --- + +export type PassRateLabel = 'INCONCLUSIVE' | 'BROKEN' | 'FLAKY' | 'FAILING' | 'PASSING'; +export type AlarmKind = 'drift' | 'rule-as-behavior' | 'regression' | 'quarantine-exit' | 'quarantine-expired' + | 'quarantine-cap' | 'quarantine-invalid'; + +/** The EVAL_POLICY fields pass-rates reads (structural, so tests can vary them). */ +export interface PassRatePolicy { + version: number; + quarantine: { entry: { rate: number; minTrials: number }; exit: { rate: number; minTrials: number }; capFraction: number; expiryWeeklyRuns: number }; + drift: { fisherAlpha: number; fisherMinPerSide: number }; +} + +export type QuarantineEntry = (typeof CASE_QUARANTINE)[string]; + +/** Tiers whose cases block a lane; quarantine applies only to them. */ +export const BLOCKING_TIERS: readonly string[] = ['gate', 'periodic']; +const QUARANTINE_FAILURE_CLASSES: readonly string[] = ['detector', 'harness', 'model-latency']; + +export interface SeriesStats { + key: string; + identity: string; + model: string; + cli: string; + policyVersion: number; + passes: number; + /** Scored trials: passed + failed (skipped trials carry no verdict). */ + trials: number; + infra: number; + interval: { lo: number; hi: number }; + firstSeen: string; + lastSeen: string; + runs: string[]; +} + +export interface CasePassRate { + case: string; + kind: EvalCaseKind; + tier: string; + quarantined: boolean; + label: PassRateLabel; + /** Manual-review acceptances: visible, never scored. */ + manualReviews: number; + current: SeriesStats | null; + previous: SeriesStats | null; + prePolicy: SeriesStats | null; + series: SeriesStats[]; + latestRun: { runId: string; passes: number; trials: number } | null; +} + +export interface Alarm { kind: AlarmKind; case: string; message: string } + +export interface PassRateReport { + policyVersion: number; + cases: CasePassRate[]; + alarms: Alarm[]; + postPolicyTrials: number; + prePolicyTrials: number; + unattributed: string[]; + errors: string[]; +} + +export interface AnalyzeOptions { + registry?: Registry; + quarantine?: Record; + policy?: PassRatePolicy; + /** Completed weekly-run timestamps in the window, for quarantine expiry. */ + weeklyRuns?: string[]; + now?: number; + unattributed?: string[]; + errors?: string[]; + manualReviews?: string[]; +} + +const at = (record: TrialRecord) => record.recorded_at ?? ''; +const runOf = (record: TrialRecord) => `${record.run_id ?? record.sha ?? 'local'}#${record.attempt}`; + +function seriesStats(key: string, records: TrialRecord[]): SeriesStats { + const scored = records.filter(record => record.outcome !== 'skipped'); + const passes = scored.filter(record => record.outcome === 'passed').length; + const times = records.map(at).sort(); + const first = records[0]!; + return { + key, identity: first.series_identity ?? 'unknown', model: first.model ?? 'unknown', cli: first.cli_version ?? 'unknown', + policyVersion: first.policy_version, passes, trials: scored.length, + infra: scored.filter(record => record.outcome === 'failed' && record.failure_class === 'infra').length, + interval: wilsonInterval(passes, scored.length), firstSeen: times[0] ?? '', lastSeen: times[times.length - 1] ?? '', + runs: [...new Set(records.map(runOf))], + }; +} + +/** Weekly runs completed after an entry's enteredAt; offline, whole weeks elapsed. */ +export function quarantineRunsSince(enteredAt: string, weeklyRuns: string[] | undefined, now: number): number { + const entered = Date.parse(enteredAt); + if (!Number.isFinite(entered)) return Number.POSITIVE_INFINITY; + if (weeklyRuns && weeklyRuns.length) return weeklyRuns.filter(time => Date.parse(time) > entered).length; + return Math.floor((now - entered) / (7 * 86_400_000)); +} + +/** + * Static CASE_QUARANTINE problems, shared by the free policy test and the + * weekly gate: an id that is not a blocking-tier E2E case, a missing field, + * a failure class outside detector / harness / model-latency (a product + * defect is fixed or named, never quarantined), a malformed or future date, + * and a tier over its cap. + */ +export function quarantinePolicyProblems(quarantine: Record, + registry: Registry = LIVE_REGISTRY, policy: PassRatePolicy = EVAL_POLICY, now = Date.now()): Alarm[] { + const problems: Alarm[] = []; + const invalid = (id: string, message: string) => problems.push({ kind: 'quarantine-invalid', case: id, message: `${id}: ${message}` }); + const perTier = new Map(); + for (const [id, entry] of Object.entries(quarantine)) { + const tier = registry.tiers[id]; + if (!tier || !(id in registry.kinds)) { invalid(id, 'CASE_QUARANTINE names no registered E2E case'); continue; } + if (!BLOCKING_TIERS.includes(tier)) invalid(id, `tier ${tier} is not blocking; only ${BLOCKING_TIERS.join(' and ')} cases are quarantined`); + for (const field of ['reason', 'failureClass', 'tracking', 'owner', 'enteredAt', 'exit'] as const) { + if (typeof entry[field] !== 'string' || !entry[field].trim()) invalid(id, `missing ${field}`); + } + if (typeof entry.reason === 'string' && entry.reason.trim().length < 40) invalid(id, 'reason must be a written diagnosis (at least 40 characters)'); + if (!QUARANTINE_FAILURE_CLASSES.includes(entry.failureClass)) { + invalid(id, `failureClass ${JSON.stringify(entry.failureClass)} is not ${QUARANTINE_FAILURE_CLASSES.join(', ')}; a product defect is fixed or named as a red, never quarantined`); + } + const entered = Date.parse(entry.enteredAt); + if (!/^\d{4}-\d{2}-\d{2}$/.test(entry.enteredAt ?? '') || !Number.isFinite(entered)) invalid(id, 'enteredAt must be YYYY-MM-DD'); + else if (entered > now) invalid(id, 'enteredAt is in the future'); + perTier.set(tier, (perTier.get(tier) ?? 0) + 1); + } + for (const [tier, count] of perTier) { + const size = Object.values(registry.tiers).filter(value => value === tier).length; + const cap = Math.floor(size * policy.quarantine.capFraction); + if (count > cap) problems.push({ kind: 'quarantine-cap', case: tier, + message: `${count} quarantined ${tier} cases exceed the ${pct(policy.quarantine.capFraction)} cap (${cap} of ${size})` }); + } + return problems; +} + +export function analyzePassRates(records: TrialRecord[], options: AnalyzeOptions = {}): PassRateReport { + const registry = options.registry ?? LIVE_REGISTRY; + const quarantine = options.quarantine ?? CASE_QUARANTINE; + const policy = options.policy ?? EVAL_POLICY; + const now = options.now ?? Date.now(); + const byCase = new Map(); + for (const record of records) { + const list = byCase.get(record.case) ?? []; + list.push(record); + byCase.set(record.case, list); + } + for (const id of options.manualReviews ?? []) if (!byCase.has(id)) byCase.set(id, []); + const cases: CasePassRate[] = []; + for (const [id, list] of [...byCase].sort(([a], [b]) => a.localeCompare(b))) { + list.sort((a, b) => at(a).localeCompare(at(b)) || runOf(a).localeCompare(runOf(b)) || a.trial - b.trial); + const groups = new Map(); + for (const record of list) { + const key = record.policy_version === 0 ? 'pre-policy' + : [record.series_identity ?? 'unknown', record.model ?? 'unknown', record.cli_version ?? 'unknown', `v${record.policy_version}`].join('|'); + const group = groups.get(key) ?? []; + group.push(record); + groups.set(key, group); + } + const series = [...groups].map(([key, group]) => seriesStats(key, group)) + .sort((a, b) => a.lastSeen.localeCompare(b.lastSeen)); + const post = series.filter(entry => entry.policyVersion !== 0); + const current = post[post.length - 1] ?? null; + const previous = post[post.length - 2] ?? null; + const scored = current ? groups.get(current.key)!.filter(record => record.outcome !== 'skipped') : []; + const latestRun = scored.length ? runOf(scored[scored.length - 1]!) : null; + const latest = scored.filter(record => runOf(record) === latestRun); + const prior = scored.filter(record => runOf(record) !== latestRun); + const priorPasses = prior.filter(record => record.outcome === 'passed').length; + const entryRate = policy.quarantine.entry.rate; + let label: PassRateLabel; + if (latest.length > 0 && latest.every(record => record.outcome === 'failed') + && prior.length > 0 && wilsonInterval(priorPasses, prior.length).lo >= entryRate) label = 'BROKEN'; + else if (!current || current.trials < policy.quarantine.entry.minTrials) label = 'INCONCLUSIVE'; + else if (current.interval.hi < entryRate) label = 'FAILING'; + else if (current.passes < current.trials && current.interval.lo < entryRate) label = 'FLAKY'; + else label = 'PASSING'; + cases.push({ + case: id, kind: registry.kinds[id] ?? list[0]!.kind, tier: caseTier(id, registry), + quarantined: id in quarantine, label, current, previous, + manualReviews: (options.manualReviews ?? []).filter(name => name === id).length, + prePolicy: series.find(entry => entry.policyVersion === 0) ?? null, series, + latestRun: latestRun ? { runId: latestRun, passes: latest.filter(record => record.outcome === 'passed').length, trials: latest.length } : null, + }); + } + + const alarms: Alarm[] = []; + const rate = (stats: SeriesStats) => stats.passes / stats.trials; + for (const entry of cases) { + const current = entry.current; + if (!current) continue; + const below = current.trials >= policy.quarantine.entry.minTrials && rate(current) < policy.quarantine.entry.rate; + if (below && !entry.quarantined && BLOCKING_TIERS.includes(entry.tier)) alarms.push({ kind: 'drift', case: entry.case, + message: `${entry.case} passes ${current.passes}/${current.trials} (below ${pct(policy.quarantine.entry.rate)} over >= ${policy.quarantine.entry.minTrials} trials): fix it, or propose a CASE_QUARANTINE entry with a written diagnosis (product defects are never quarantined)` }); + if (below && entry.kind === 'rule') alarms.push({ kind: 'rule-as-behavior', case: entry.case, + message: `${entry.case}: rule case behaving like behavior (${current.passes}/${current.trials}): fix or reclassify` }); + if (entry.quarantined && current.trials >= policy.quarantine.exit.minTrials && rate(current) >= policy.quarantine.exit.rate) { + alarms.push({ kind: 'quarantine-exit', case: entry.case, + message: `${entry.case} passes ${current.passes}/${current.trials} (>= ${pct(policy.quarantine.exit.rate)}): remove its CASE_QUARANTINE entry` }); + } + } + const tested = cases.filter(entry => BLOCKING_TIERS.includes(entry.tier) && entry.current && entry.previous + && entry.current.trials >= policy.drift.fisherMinPerSide && entry.previous.trials >= policy.drift.fisherMinPerSide); + const pValues = tested.map(entry => fisherOneSidedLower(entry.current!.passes, entry.current!.trials, entry.previous!.passes, entry.previous!.trials)); + for (const index of holmRejections(pValues, policy.drift.fisherAlpha)) { + const entry = tested[index]!; + alarms.push({ kind: 'regression', case: entry.case, + message: `${entry.case}: current identity ${entry.current!.passes}/${entry.current!.trials} is significantly below the previous ${entry.previous!.passes}/${entry.previous!.trials} (one-sided Fisher p=${pValues[index]!.toFixed(4)}, Holm over ${tested.length} cases)` }); + } + for (const [id, entry] of Object.entries(quarantine)) { + const runs = quarantineRunsSince(entry.enteredAt, options.weeklyRuns, now); + if (runs >= policy.quarantine.expiryWeeklyRuns) alarms.push({ kind: 'quarantine-expired', case: id, + message: `${id}: entered ${runs} weekly runs ago (limit ${policy.quarantine.expiryWeeklyRuns}): fix it, name it as a red, or re-diagnose with fresh evidence` }); + } + alarms.push(...quarantinePolicyProblems(quarantine, registry, policy, now)); + + const post = records.filter(record => record.policy_version !== 0).length; + return { policyVersion: policy.version, cases, alarms, postPolicyTrials: post, prePolicyTrials: records.length - post, + unattributed: options.unattributed ?? [], errors: options.errors ?? [] }; +} + +function pct(value: number): string { return `${Math.round(value * 1000) / 10}%`; } + +function formatStats(stats: SeriesStats | null): string { + if (!stats) return '-'; + return `${stats.passes}/${stats.trials} [${pct(stats.interval.lo)}–${pct(stats.interval.hi)}]${stats.infra ? ` (${stats.infra} infra)` : ''}`; +} + +export function formatPassRates(report: PassRateReport, options: { caseFilter?: string } = {}): string { + const lines: string[] = []; + lines.push(`pass-rates: policy v${report.policyVersion}, ${report.postPolicyTrials} post-policy trial(s), ${report.prePolicyTrials} pre-policy (display only)`); + if (report.postPolicyTrials === 0) lines.push(' no post-policy trials yet: every series starts INCONCLUSIVE'); + const cases = report.cases.filter(entry => !options.caseFilter || entry.case === options.caseFilter); + lines.push(' label kind tier current series pre-policy manual case'); + for (const entry of cases) { + const group = entry.current ? ` ${entry.current.model} / ${entry.current.cli}` : ''; + const reset = entry.previous ? ' (baseline reset)' : ''; + lines.push(` ${entry.label.padEnd(12)} ${entry.kind.padEnd(8)} ${entry.tier.padEnd(8)} ${formatStats(entry.current).padEnd(29)} ` + + `${formatStats(entry.prePolicy).padEnd(18)} ${String(entry.manualReviews).padStart(6)} ${entry.case}${entry.quarantined ? ' [quarantined]' : ''}${group}${reset}`); + } + if (report.unattributed.length) lines.push(` unattributed records (${report.unattributed.length}, never guessed): ${report.unattributed.slice(0, 20).join(', ')}`); + if (report.errors.length) lines.push(` rejected ${report.errors.length} invalid trial line(s): ${report.errors.slice(0, 5).join('; ')}`); + if (report.alarms.length) { + lines.push(`ACTION REQUIRED (${report.alarms.length}):`); + for (const alarm of report.alarms) lines.push(` [${alarm.kind}] ${alarm.message}`); + } + return lines.join('\n'); +} + +// --- GitHub history --- + +export interface WeeklyRun { id: number; attempt: number; sha: string; branch: string; createdAt: string } +export interface RunArtifact { id: number; name: string; size: number } + +/** The GitHub calls pass-rates makes; injectable so the free tests never touch the network. */ +export interface HistoryFetcher { + listRuns(repo: string, workflow: string, branch: string, limit: number): WeeklyRun[]; + listArtifacts(repo: string, runId: number): RunArtifact[]; + downloadZip(repo: string, artifactId: number, destination: string): void; +} + +function gh(args: string[]): Buffer { + const result = spawnSync('gh', args, { timeout: 300_000, maxBuffer: 256 * 1024 * 1024 }); + if (result.status !== 0) throw new Error(`gh ${args.slice(0, 2).join(' ')} failed: ${String(result.stderr || result.error || '').trim()}`); + return result.stdout; +} + +const jsonLines = (buffer: Buffer): T[] => buffer.toString('utf8').split('\n').filter(Boolean).map(line => JSON.parse(line) as T); + +export const GH_HISTORY: HistoryFetcher = { + listRuns: (repo, workflow, branch, limit) => jsonLines(gh(['api', + `repos/${repo}/actions/workflows/${workflow}/runs?branch=${encodeURIComponent(branch)}&status=completed&per_page=${limit}`, + '--jq', '.workflow_runs[] | {id, attempt: .run_attempt, sha: .head_sha, branch: .head_branch, createdAt: .created_at}'])), + listArtifacts: (repo, runId) => jsonLines(gh(['api', `repos/${repo}/actions/runs/${runId}/artifacts?per_page=100`, + '--paginate', '--jq', '.artifacts[] | select(.expired | not) | {id, name, size: .size_in_bytes}'])), + downloadZip: (repo, artifactId, destination) => fs.writeFileSync(destination, gh(['api', `repos/${repo}/actions/artifacts/${artifactId}/zip`])), +}; + +/** The last `limit` completed runs of `workflow` on each branch, newest first, deduplicated. */ +export function listWeeklyRuns(opts: { repo: string; workflow: string; branches: string[]; limit: number; fetcher?: HistoryFetcher }): WeeklyRun[] { + const fetcher = opts.fetcher ?? GH_HISTORY; + const runs = new Map(); + for (const branch of opts.branches) for (const run of fetcher.listRuns(opts.repo, opts.workflow, branch, opts.limit)) runs.set(run.id, run); + return [...runs.values()].sort((a, b) => b.createdAt.localeCompare(a.createdAt)); +} + +/** + * Download the artifacts of one run whose names match into a per-run cache + * directory (reused on later calls) and return the extracted directories. + * Oversized or oddly named artifacts are skipped: downloads are data only. + */ +export function downloadRunArtifacts(opts: { repo: string; run: WeeklyRun; match: (name: string) => boolean; cacheDir: string; + fetcher?: HistoryFetcher; maxBytes?: number }): string[] { + const fetcher = opts.fetcher ?? GH_HISTORY; + const dirs: string[] = []; + for (const artifact of fetcher.listArtifacts(opts.repo, opts.run.id)) { + if (!opts.match(artifact.name) || !/^[A-Za-z0-9._-]+$/.test(artifact.name)) continue; + if (artifact.size > (opts.maxBytes ?? TRIAL_OUTCOMES_MAX_BYTES)) continue; + const dir = path.join(opts.cacheDir, `${opts.run.id}`, artifact.name); + if (!fs.existsSync(path.join(dir, '.complete'))) { + fs.rmSync(dir, { recursive: true, force: true }); + fs.mkdirSync(dir, { recursive: true }); + const zip = path.join(dir, 'artifact.zip'); + fetcher.downloadZip(opts.repo, artifact.id, zip); + const unzip = spawnSync('unzip', ['-o', '-q', zip, '-d', dir], { timeout: 120_000 }); + if (unzip.status !== 0) throw new Error(`unzip failed for ${artifact.name}: ${String(unzip.stderr || unzip.error || '')}`); + fs.rmSync(zip, { force: true }); + fs.writeFileSync(path.join(dir, '.complete'), ''); + } + dirs.push(dir); + } + return dirs; +} + +function gitOutput(args: string[]): string | null { + const result = spawnSync('git', args, { encoding: 'utf8', timeout: 5_000 }); + return result.status === 0 ? result.stdout.trim() : null; +} + +function repoSlug(): string { + const url = gitOutput(['remote', 'get-url', 'origin']) ?? ''; + return url.match(/[:/]([^/:]+\/[^/]+?)(?:\.git)?$/)?.[1] ?? 'garrytan/gstack'; +} + if (import.meta.main) { const argv = process.argv.slice(2); - const dirFlag = argv.indexOf('--dir'); - const dir = dirFlag !== -1 ? argv[dirFlag + 1] : getProjectEvalDir(); + const flag = (name: string) => { const index = argv.indexOf(name); return index === -1 ? undefined : argv[index + 1]; }; + const dirs = argv.flatMap((arg, index) => arg === '--dir' && argv[index + 1] ? [argv[index + 1]!] : []); const asJson = argv.includes('--json'); - const sinceFlag = argv.indexOf('--since-days'); - const sinceDays = sinceFlag !== -1 ? Number(argv[sinceFlag + 1]) || 60 : 60; + const gate = argv.includes('--gate'); + const backfill = argv.includes('--backfill'); + const caseFilter = flag('--case'); + const runsLimit = Number(flag('--runs')) || 10; + const sinceDays = Number(flag('--since-days')) || 60; + const repo = flag('--repo') ?? repoSlug(); + const workflow = flag('--workflow') ?? 'evals-periodic.yml'; + const branch = flag('--branch') ?? gitOutput(['rev-parse', '--abbrev-ref', 'HEAD']) ?? 'main'; - const files = collectEvalFiles(dir, sinceDays); - const series = [...aggregate(files).values()] - .sort((a, b) => b.retriedPasses - a.retriedPasses || (b.fails / Math.max(1, b.runs)) - (a.fails / Math.max(1, a.runs))); - const ledger = readFreeLedger(); + const records: TrialRecord[] = []; + const unattributed = new Set(); + const errors: string[] = []; + let historyError: string | null = null; + let weeklyRuns: string[] | undefined; - if (asJson) { - console.log(JSON.stringify({ dir, runsScanned: files.length, tests: series, freeLedger: ledger }, null, 2)); + const manualReviews: string[] = []; + const importDir = (dir: string, run: { run_id: string; sha?: string; timestamp?: string } | undefined, legacyDays: number) => { + const trials = readTrialOutcomeDir(dir); + records.push(...trials.records); + errors.push(...trials.errors); + const legacy = backfillEvalFiles(collectEvalFiles(dir, legacyDays), run); + records.push(...legacy.records); + manualReviews.push(...legacy.manualReviews); + legacy.unattributed.forEach(name => unattributed.add(name)); + }; + + if (dirs.length) { + for (const dir of dirs) importDir(dir, undefined, sinceDays); } else { - console.log(`flake-rank: ${files.length} finalized run file(s) under ${dir}`); - const flaky = series.filter((s) => s.retriedPasses > 0 || s.fails > 0 || s.manualAccepted > 0); - if (flaky.length === 0) { - console.log(' no retried passes and no failures recorded — clean series'); - } else { - console.log(' retries fails/runs manual avg-dur test'); - for (const s of flaky.slice(0, 30)) { - console.log(` ${String(s.retriedPasses).padStart(7)} ${String(s.fails).padStart(5)}/${String(s.runs).padEnd(6)} ` - + `${String(s.manualAccepted).padStart(6)} ${Math.round(s.totalDurationMs / s.totalAttempts / 1000).toString().padStart(5)}s ${s.name}`); + try { + const runs = listWeeklyRuns({ repo, workflow, branches: [...new Set([branch, 'main'])], limit: runsLimit }); + weeklyRuns = runs.map(run => run.createdAt); + const cacheDir = path.join(os.homedir(), '.gstack', 'eval-pass-rates-cache', repo.replace('/', '-')); + const match = backfill + ? (name: string) => name.startsWith('trial-outcomes') || /^(paid-slice-\d+|gate-census-\d+)$/.test(name) + : (name: string) => name.startsWith('trial-outcomes'); + for (const run of runs) { + const dirsForRun = downloadRunArtifacts({ repo, run, match, cacheDir, maxBytes: backfill ? 64 * 1024 * 1024 : undefined }); + for (const dir of dirsForRun) importDir(dir, { run_id: `${run.id}`, sha: run.sha, timestamp: run.createdAt }, 3650); } + } catch (error) { + historyError = error instanceof Error ? error.message : String(error); } + } + + const report = analyzePassRates(records, { weeklyRuns, unattributed: [...unattributed].sort(), errors, manualReviews }); + const ledger = readFreeLedger(); + if (asJson) { + console.log(JSON.stringify({ repo, workflow, branch, dirs, historyError, ...report, freeLedger: ledger }, null, 2)); + } else { + if (historyError) console.log(`pass-rates: history unavailable (${historyError}); every label below is INCONCLUSIVE`); + console.log(formatPassRates(report, { caseFilter })); if (ledger.length > 0) { const byFile = new Map(); for (const e of ledger) byFile.set(e.file, (byFile.get(e.file) ?? 0) + 1); @@ -147,4 +698,5 @@ if (import.meta.main) { } } } + if (gate && (historyError || report.alarms.length)) process.exit(1); } diff --git a/test/eval-flake-rank.test.ts b/test/eval-flake-rank.test.ts index 198a64d54..0eaa1ddab 100644 --- a/test/eval-flake-rank.test.ts +++ b/test/eval-flake-rank.test.ts @@ -13,6 +13,13 @@ import * as path from 'node:path'; import { spawnSync } from 'node:child_process'; import { aggregate, collectEvalFiles } from '../scripts/eval-flake-rank'; import { manualReviewFixture } from './helpers/manual-judge-review-fixture'; +import { + analyzePassRates, attributeLegacyRecord, backfillEvalFiles, caseSeriesIdentities, downloadRunArtifacts, fisherOneSidedLower, + formatPassRates, holmRejections, listWeeklyRuns, quarantinePolicyProblems, quarantineRunsSince, readTrialOutcomeDir, + wilsonInterval, type HistoryFetcher, type PassRatePolicy, type QuarantineEntry, type Registry, type TrialRecord, +} from '../scripts/eval-flake-rank'; +import { EVAL_POLICY } from './helpers/periodic-exclude-data'; +import { TRIAL_OUTCOME_SCHEMA, formatTrialOutcomes } from './helpers/eval-store'; const entry = (name: string, passed: boolean, attempt: number) => ({ name, suite: 's', tier: 'e2e', passed, attempt, duration_ms: 1000, cost_usd: 0.1, @@ -39,9 +46,10 @@ describe('eval-flake-rank aggregate', () => { const display = spawnSync(process.execPath, [path.resolve(import.meta.dir, '../scripts/eval-flake-rank.ts'), '--dir', dir], { encoding: 'utf8', timeout: 10_000 }); expect(display.status, display.stderr).toBe(0); - expect(display.stdout).toContain('fails/runs manual'); - expect(display.stdout).toContain('0/1'); - expect(display.stdout).toContain(manual.name); + // pass-rates view: the prior automated pass is the one scored pre-policy + // trial; the manual acceptance is counted in its own column, never scored. + expect(display.stdout).toContain('pre-policy manual case'); + expect(display.stdout).toMatch(new RegExp(`1/1 \\[[^\\]]+\\]\\s+1 ${manual.name.replace(/[.*+?^${}()|[\]\\/]/g, '\\$&')}`)); fs.writeFileSync(path.join(dir, 'invalid-retry.json'), run([ { ...ordinary, attempt: 1 }, { ...manual, attempt: 2 }, ])); @@ -97,3 +105,298 @@ describe('eval-flake-rank aggregate', () => { fs.rmSync(dir, { recursive: true, force: true }); }); }); + +// --- pass-rates --- + +const registry: Registry = { + kinds: { 'rule-a': 'rule', 'beh-b': 'behavior', 'gate-c': 'rule', 'mar-d': 'rule', 'judge one': 'judge', + ...Object.fromEntries(Array.from({ length: 18 }, (_, i) => [`filler-${i}`, 'rule'])) }, + tiers: { 'rule-a': 'periodic', 'beh-b': 'periodic', 'gate-c': 'gate', 'mar-d': 'marathon', + ...Object.fromEntries(Array.from({ length: 18 }, (_, i) => [`filler-${i}`, i < 9 ? 'gate' : 'periodic'])) }, + touchfiles: { 'rule-a': ['test/skill-e2e-a.test.ts', 'a/**'], 'beh-b': ['test/skill-e2e-b.test.ts', 'b/**'], + 'gate-c': ['test/skill-e2e-shared.test.ts'], 'mar-d': ['test/skill-e2e-shared.test.ts'] }, + judgeTouchfiles: { 'judge one': ['j/SKILL.md'] }, + globals: ['harness/**'], + testNames: { 'gate-c': '/gate c labeled' }, +}; + +let clock = 0; +function trial(id: string, outcome: 'passed' | 'failed' | 'skipped', extra: Partial = {}): TrialRecord { + clock += 1; + return { + schema: TRIAL_OUTCOME_SCHEMA, case: id, file: 'test/x.test.ts', tier: registry.tiers[id] ?? 'judge', + kind: registry.kinds[id]!, trial: 1, panel: { n: 1, k: 1 }, attempt: 1, outcome, + ...(outcome === 'failed' ? { failure_class: 'assertion' as const } : {}), + duration_ms: 1, cost_usd: 0, model: 'model-x', cli_version: '2.1.284', policy_version: 1, quarantined: false, + execution: 'executed', source: 'shard', run_id: `run-${clock}`, recorded_at: new Date(Date.UTC(2026, 9, 1) + clock * 60_000).toISOString(), + series_identity: 'id-1', ...extra, + }; +} +const many = (id: string, passes: number, fails: number, extra: Partial = {}) => + [...Array.from({ length: passes }, () => trial(id, 'passed', extra)), ...Array.from({ length: fails }, () => trial(id, 'failed', extra))]; +const analyze = (records: TrialRecord[], quarantine: Record = {}, extra = {}) => + analyzePassRates(records, { registry, quarantine, now: Date.UTC(2026, 9, 2), ...extra }); +const qEntry = (overrides: Partial = {}): QuarantineEntry => ({ + reason: 'Detector graded the posture wording; 7 of 10 fresh trials failed only the regex, transcripts attached.', + failureClass: 'detector', tracking: 'TODOS.md "x"', owner: 'garrytan', enteredAt: '2026-09-29', + exit: '>= 97% over >= 10 trials on the current identity', ...overrides, +}); + +describe('pass-rates statistics', () => { + test('Wilson bounds match the documented policy arithmetic', () => { + expect(wilsonInterval(10, 10).lo).toBeCloseTo(0.7225, 4); + expect(wilsonInterval(6, 6).lo).toBeCloseTo(0.6097, 4); + expect(wilsonInterval(125, 125).lo).toBeCloseTo(0.9702, 4); + expect(wilsonInterval(10, 10).hi).toBe(1); + expect(wilsonInterval(0, 0)).toEqual({ lo: 0, hi: 1 }); + const mid = wilsonInterval(7, 10); + expect(mid.lo).toBeGreaterThan(0.39); expect(mid.hi).toBeLessThan(0.9); + }); + + test('one-sided Fisher exact matches a known table and is one-sided', () => { + expect(fisherOneSidedLower(4, 6, 6, 6)).toBeCloseTo(0.227272727, 8); + expect(fisherOneSidedLower(6, 6, 4, 6)).toBe(1); + expect(fisherOneSidedLower(0, 10, 10, 10)).toBeLessThan(1e-4); + }); + + test('Holm rejects step-down and stops at the first non-rejection', () => { + expect([...holmRejections([0.001, 0.02, 0.04], 0.05)].sort()).toEqual([0, 1, 2]); + expect([...holmRejections([0.001, 0.03, 0.04], 0.05)]).toEqual([0]); + expect([...holmRejections([0.03, 0.04], 0.05)]).toEqual([]); + expect([...holmRejections([], 0.05)]).toEqual([]); + }); +}); + +describe('pass-rates labels', () => { + test('thin history is INCONCLUSIVE, and after this PR every series starts there', () => { + const report = analyze(many('rule-a', 9, 0)); + expect(report.cases[0]).toMatchObject({ case: 'rule-a', label: 'INCONCLUSIVE' }); + expect(formatPassRates(analyze([]))).toContain('every series starts INCONCLUSIVE'); + }); + + test('PASSING, FLAKY and FAILING come from the interval against the entry rate', () => { + expect(analyze(many('rule-a', 12, 0)).cases[0]!.label).toBe('PASSING'); + expect(analyze(many('beh-b', 10, 1)).cases[0]!.label).toBe('FLAKY'); + expect(analyze(many('beh-b', 2, 10)).cases[0]!.label).toBe('FAILING'); + }); + + test('BROKEN: the latest run is 0/n after a prior interval at or above the entry rate', () => { + const prior = many('beh-b', 80, 0, { run_id: 'old' }); + const latest = [1, 2, 3].map(n => trial('beh-b', 'failed', { run_id: 'new', trial: n, panel: { n: 3, k: 2 } })); + expect(analyze([...prior, ...latest]).cases[0]!.label).toBe('BROKEN'); + }); + + test('skipped trials carry no verdict; infra failures count as failed trials', () => { + const stats = analyze([...many('rule-a', 10, 0), trial('rule-a', 'skipped'), + trial('rule-a', 'failed', { failure_class: 'infra' })]).cases[0]!.current!; + expect(stats).toMatchObject({ passes: 10, trials: 11, infra: 1 }); + }); + + test('a new identity, model or CLI starts a new series; earlier series stay visible', () => { + const report = analyze([...many('rule-a', 10, 0), ...many('rule-a', 3, 0, { series_identity: 'id-2' }), + ...many('rule-a', 2, 0, { series_identity: 'id-2', cli_version: '2.1.285' })]); + const c = report.cases[0]!; + expect(c.series).toHaveLength(3); + expect(c.current).toMatchObject({ identity: 'id-2', cli: '2.1.285', trials: 2 }); + expect(c.previous).toMatchObject({ identity: 'id-2', cli: '2.1.284', trials: 3 }); + expect(c.label).toBe('INCONCLUSIVE'); + }); +}); + +describe('pass-rates alarms count post-policy trials of the current series only', () => { + test('backfilled pre-policy failures are displayed but never alarm', () => { + const report = analyze(many('rule-a', 2, 20, { policy_version: 0, source: 'backfill' })); + expect(report.alarms).toEqual([]); + expect(report.cases[0]!.prePolicy).toMatchObject({ passes: 2, trials: 22 }); + expect(report.cases[0]!.label).toBe('INCONCLUSIVE'); + }); + + test('drift proposes quarantine for a blocking case below the entry rule; a rule case is flagged as behaving like behavior', () => { + const kinds = analyze([...many('rule-a', 8, 2), ...many('beh-b', 8, 2), ...many('mar-d', 0, 10)]).alarms.map(a => `${a.kind}:${a.case}`); + expect([...kinds].sort()).toEqual(['drift:beh-b', 'drift:rule-a', 'rule-as-behavior:mar-d', 'rule-as-behavior:rule-a']); + expect(analyze(many('rule-a', 19, 1)).alarms).toEqual([]); + }); + + test('the Fisher regression alarm needs the minimum trials on both sides', () => { + const old = many('gate-c', 6, 0, { series_identity: 'old' }); + const fresh = many('gate-c', 0, 6, { series_identity: 'new' }); + expect(analyze([...old, ...fresh]).alarms.map(a => a.kind)).toContain('regression'); + expect(analyze([...old, ...fresh.slice(0, 5)]).alarms.map(a => a.kind)).not.toContain('regression'); + }); + + test('quarantine exit, expiry and cap', () => { + const exit = analyze(many('beh-b', 10, 0), { 'beh-b': qEntry() }).alarms.map(a => a.kind); + expect(exit).toContain('quarantine-exit'); + expect(exit).not.toContain('drift'); + const weekly = Array.from({ length: 8 }, (_, i) => new Date(Date.UTC(2026, 8, 30) + i * 7 * 86_400_000).toISOString()); + expect(analyze([], { 'beh-b': qEntry() }, { weeklyRuns: weekly }).alarms.map(a => a.kind)).toContain('quarantine-expired'); + expect(analyze([], { 'beh-b': qEntry() }, { weeklyRuns: weekly.slice(0, 7) }).alarms.map(a => a.kind)).not.toContain('quarantine-expired'); + expect(quarantineRunsSince('2026-09-01', undefined, Date.UTC(2026, 9, 27))).toBe(8); + expect(quarantineRunsSince('not a date', undefined, 0)).toBe(Number.POSITIVE_INFINITY); + }); +}); + +describe('quarantine policy', () => { + const policy: PassRatePolicy = EVAL_POLICY; + const now = Date.UTC(2026, 9, 2); + test('a valid entry has no problems', () => { + expect(quarantinePolicyProblems({ 'beh-b': qEntry() }, registry, policy, now)).toEqual([]); + }); + + test('a product defect, a missing diagnosis or field, a bad date or a non-blocking case is rejected', () => { + const problems = (quarantine: Record) => quarantinePolicyProblems(quarantine, registry, policy, now).map(p => p.message); + expect(problems({ 'beh-b': qEntry({ failureClass: 'product' as QuarantineEntry['failureClass'] }) }).join()).toContain('never quarantined'); + expect(problems({ 'beh-b': qEntry({ reason: 'flaky' }) }).join()).toContain('written diagnosis'); + expect(problems({ 'beh-b': qEntry({ owner: ' ' }) }).join()).toContain('missing owner'); + expect(problems({ 'beh-b': qEntry({ enteredAt: '09/29/2026' }) }).join()).toContain('YYYY-MM-DD'); + expect(problems({ 'beh-b': qEntry({ enteredAt: '2027-01-01' }) }).join()).toContain('future'); + expect(problems({ 'mar-d': qEntry() }).join()).toContain('not blocking'); + expect(problems({ 'judge one': qEntry() }).join()).toContain('no registered E2E case'); + expect(problems({ ghost: qEntry() }).join()).toContain('no registered E2E case'); + }); + + test('at most 10% of a tier may be quarantined', () => { + // 11 periodic cases in the fixture registry: the cap is 1. + expect(quarantinePolicyProblems({ 'beh-b': qEntry() }, registry, policy, now)).toEqual([]); + const over = quarantinePolicyProblems({ 'beh-b': qEntry(), 'rule-a': qEntry() }, registry, policy, now); + expect(over.map(p => p.kind)).toEqual(['quarantine-cap']); + expect(over[0]!.message).toContain('2 quarantined periodic cases exceed the 10% cap (1 of 11)'); + }); +}); + +describe('pass-rates inputs', () => { + test('trial-outcomes JSONL is schema-validated; invalid lines are reported, never guessed', () => { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'passrates-')); + const valid = trial('rule-a', 'passed'); + fs.mkdirSync(path.join(dir, 'nested')); + fs.writeFileSync(path.join(dir, 'nested', 'trial-outcomes.jsonl'), formatTrialOutcomes([valid]) + '{"schema":"other"}\nnot json\n'); + fs.writeFileSync(path.join(dir, 'unrelated.jsonl'), formatTrialOutcomes([valid])); + const read = readTrialOutcomeDir(dir); + expect(read.records).toHaveLength(1); + expect(read.records[0]).toMatchObject({ case: 'rule-a', series_identity: 'id-1' }); + expect(read.errors).toHaveLength(2); + fs.rmSync(dir, { recursive: true, force: true }); + }); + + test('legacy records attribute by shard suffix, id, label or single-owner file, else stay unattributed', () => { + expect(attributeLegacyRecord('/anything', 'skill-e2e-b--beh-b', registry)).toBe('beh-b'); + expect(attributeLegacyRecord('rule-a', 'skill-e2e-zzz', registry)).toBe('rule-a'); + expect(attributeLegacyRecord('/gate c labeled', undefined, registry)).toBe('gate-c'); + expect(attributeLegacyRecord('/a display name', 'skill-e2e-a', registry)).toBe('rule-a'); + expect(attributeLegacyRecord('/shared display', 'skill-e2e-shared', registry)).toBeNull(); + }); + + test('backfill keeps only first attempts, defaults a missing attempt to 1, and labels records pre-policy', () => { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'passrates-backfill-')); + fs.writeFileSync(path.join(dir, 'run.json'), run([ + { ...entry_('rule-a', false, 1), exit_reason: 'timeout' }, entry_('rule-a', true, 2), + { name: 'beh-b', suite: 's', tier: 'e2e', passed: true, duration_ms: 1, cost_usd: 0 }, + entry_('/unknown display', true, 1), + ], { shard: 'skill-e2e-zzz' })); + const { records, unattributed } = backfillEvalFiles(collectEvalFiles(dir), { run_id: '42', sha: 'abc' }, registry); + expect(records.map(r => [r.case, r.outcome, r.failure_class, r.policy_version, r.source, r.run_id])) + .toEqual([['rule-a', 'failed', 'timeout', 0, 'backfill', '42'], ['beh-b', 'passed', undefined, 0, 'backfill', '42']]); + expect(formatTrialOutcomes(records)).toContain(TRIAL_OUTCOME_SCHEMA); + expect(unattributed).toEqual(['/unknown display']); + fs.rmSync(dir, { recursive: true, force: true }); + }); + + test('series identity follows the case\'s own touchfiles, not GLOBAL_TOUCHFILES', () => { + const root = fs.mkdtempSync(path.join(os.tmpdir(), 'passrates-series-')); + const git = (...args: string[]) => spawnSync('git', args, { cwd: root, encoding: 'utf8', timeout: 10_000 }); + for (const [file, body] of [['a/x.ts', '1'], ['b/y.ts', '1'], ['harness/run.ts', '1'], ['test/skill-e2e-a.test.ts', '1']] as const) { + fs.mkdirSync(path.join(root, path.dirname(file)), { recursive: true }); + fs.writeFileSync(path.join(root, file), body); + } + const snapshot = () => { expect(git('add', '-A').status).toBe(0); return caseSeriesIdentities(['rule-a', 'beh-b'], root, registry); }; + expect(git('init', '-q').status).toBe(0); + const first = snapshot(); + expect(first['rule-a']).not.toBe(first['beh-b']); + fs.writeFileSync(path.join(root, 'harness/run.ts'), '2'); + expect(snapshot()).toEqual(first); + fs.writeFileSync(path.join(root, 'a/x.ts'), '2'); + const next = snapshot(); + expect(next['rule-a']).not.toBe(first['rule-a']); + expect(next['beh-b']).toBe(first['beh-b']); + fs.rmSync(root, { recursive: true, force: true }); + }); +}); + +describe('pass-rates history fetch (injected, no network)', () => { + function storedZip(files: Record): Buffer { + const locals: Buffer[] = [], centrals: Buffer[] = []; + let offset = 0; + for (const [name, text] of Object.entries(files)) { + const data = Buffer.from(text), fileName = Buffer.from(name), crc = Bun.hash.crc32(data) >>> 0; + const local = Buffer.alloc(30); local.writeUInt32LE(0x04034b50, 0); local.writeUInt16LE(20, 4); + local.writeUInt32LE(crc, 14); local.writeUInt32LE(data.length, 18); local.writeUInt32LE(data.length, 22); local.writeUInt16LE(fileName.length, 26); + const central = Buffer.alloc(46); central.writeUInt32LE(0x02014b50, 0); central.writeUInt16LE(20, 4); central.writeUInt16LE(20, 6); + central.writeUInt32LE(crc, 16); central.writeUInt32LE(data.length, 20); central.writeUInt32LE(data.length, 24); + central.writeUInt16LE(fileName.length, 28); central.writeUInt32LE(offset, 42); + locals.push(local, fileName, data); centrals.push(central, fileName); + offset += 30 + fileName.length + data.length; + } + const size = centrals.reduce((sum, b) => sum + b.length, 0); + const end = Buffer.alloc(22); end.writeUInt32LE(0x06054b50, 0); end.writeUInt16LE(Object.keys(files).length, 8); + end.writeUInt16LE(Object.keys(files).length, 10); end.writeUInt32LE(size, 12); end.writeUInt32LE(offset, 16); + return Buffer.concat([...locals, ...centrals, end]); + } + + test('lists runs per branch, deduplicated and newest first', () => { + const fetcher: HistoryFetcher = { + listRuns: (_repo, _workflow, branch) => branch === 'main' + ? [{ id: 1, attempt: 1, sha: 'a', branch, createdAt: '2026-09-01T00:00:00Z' }, { id: 3, attempt: 1, sha: 'c', branch, createdAt: '2026-09-15T00:00:00Z' }] + : [{ id: 3, attempt: 1, sha: 'c', branch, createdAt: '2026-09-15T00:00:00Z' }, { id: 2, attempt: 2, sha: 'b', branch, createdAt: '2026-09-08T00:00:00Z' }], + listArtifacts: () => [], downloadZip: () => { throw new Error('unused'); }, + }; + expect(listWeeklyRuns({ repo: 'o/r', workflow: 'evals-periodic.yml', branches: ['feature', 'main'], limit: 10, fetcher }).map(r => r.id)).toEqual([3, 2, 1]); + }); + + test('downloads only matching, bounded artifacts once, and caches them', () => { + const cacheDir = fs.mkdtempSync(path.join(os.tmpdir(), 'passrates-cache-')); + const downloads: number[] = []; + const fetcher: HistoryFetcher = { + listRuns: () => [], + listArtifacts: () => [{ id: 10, name: 'trial-outcomes-gate', size: 100 }, { id: 11, name: 'paid-slice-1', size: 100 }, + { id: 12, name: 'trial-outcomes-huge', size: 10 ** 9 }, { id: 13, name: 'trial-outcomes/../escape', size: 1 }], + downloadZip: (_repo, id, destination) => { downloads.push(id); fs.writeFileSync(destination, storedZip({ 'trial-outcomes.jsonl': formatTrialOutcomes([trial('rule-a', 'passed')]) })); }, + }; + const options = { repo: 'o/r', run: { id: 7, attempt: 1, sha: 's', branch: 'main', createdAt: '' }, cacheDir, fetcher, + match: (name: string) => name.startsWith('trial-outcomes') }; + const dirs = downloadRunArtifacts(options); + expect(downloads).toEqual([10]); + expect(dirs).toHaveLength(1); + expect(readTrialOutcomeDir(dirs[0]!).records.map(r => r.case)).toEqual(['rule-a']); + expect(downloadRunArtifacts(options)).toEqual(dirs); + expect(downloads).toEqual([10]); + fs.rmSync(cacheDir, { recursive: true, force: true }); + }); +}); + +describe('pass-rates CLI', () => { + const cli = (args: string[]) => spawnSync(process.execPath, [path.resolve(import.meta.dir, '../scripts/eval-flake-rank.ts'), ...args], + { encoding: 'utf8', timeout: 20_000 }); + + test('--dir prints per-case pass rates; --gate fails only on ACTION REQUIRED', () => { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'passrates-cli-')); + const id = 'plan-ceo-review-format-mode'; + const records = Array.from({ length: 12 }, (_, i) => ({ ...trial(id, i < 11 ? 'passed' : 'failed'), kind: 'behavior' as const, tier: 'periodic' })); + fs.writeFileSync(path.join(dir, 'trial-outcomes.jsonl'), formatTrialOutcomes(records)); + const shown = cli(['--dir', dir, '--case', id]); + expect(shown.status, shown.stderr).toBe(0); + expect(shown.stdout).toContain(`11/12 [`); + expect(shown.stdout).toMatch(new RegExp(`FLAKY\\s+behavior\\s+periodic.*${id}`)); + expect(shown.stdout).toContain('ACTION REQUIRED'); + expect(shown.stdout).toContain(`[drift] ${id} passes 11/12`); + expect(cli(['--dir', dir, '--gate']).status).toBe(1); + fs.writeFileSync(path.join(dir, 'trial-outcomes.jsonl'), formatTrialOutcomes(records.slice(0, 11))); + const clean = cli(['--dir', dir, '--gate', '--json']); + expect(clean.status, clean.stdout).toBe(0); + expect(JSON.parse(clean.stdout).cases[0]).toMatchObject({ case: id, label: 'PASSING', current: { passes: 11, trials: 11 } }); + fs.rmSync(dir, { recursive: true, force: true }); + }); +}); + +function entry_(name: string, passed: boolean, attempt: number) { + return { name, suite: 's', tier: 'e2e', passed, attempt, duration_ms: 1000, cost_usd: 0.1 }; +} diff --git a/test/helpers/periodic-exclude-data.ts b/test/helpers/periodic-exclude-data.ts index 5628e13d7..638c66d18 100644 --- a/test/helpers/periodic-exclude-data.ts +++ b/test/helpers/periodic-exclude-data.ts @@ -106,14 +106,18 @@ export const EVAL_POLICY = { * failures (a product defect is never quarantined), and unchanged case * touchfiles in the change that adds it. Pinned by * test/periodic-exclude-policy.test.ts. - * reason - the written diagnosis - * tracking - issue or TODOS pointer - * owner - who removes it - * enteredAt - ISO date the entry landed (expiry counts weekly runs from here) - * exit - the measurable exit condition + * reason - the written diagnosis, with the pass-rate evidence + * failureClass - what the diagnosis found; a product defect has no class here + * tracking - issue or TODOS pointer + * owner - who removes it + * enteredAt - YYYY-MM-DD the entry landed (expiry counts weekly runs from here) + * exit - the measurable exit condition + * At most EVAL_POLICY.quarantine.capFraction of a tier's cases (gate and + * periodic are the blocking tiers) may be quarantined at once. */ export const CASE_QUARANTINE: Record { } }); }); + +describe('eval verdict policy (pre-registered)', () => { + test('EVAL_POLICY carries exactly the approved constants; a change needs re-approval and a version bump', () => { + expect(EVAL_POLICY).toEqual({ + version: 1, + panel: { n: 3, k: 2 }, + quarantine: { entry: { rate: 0.95, minTrials: 10 }, exit: { rate: 0.97, minTrials: 10 }, capFraction: 0.10, expiryWeeklyRuns: 8 }, + judge: { samples: 3 }, + drift: { fisherAlpha: 0.05, fisherMinPerSide: 6 }, + infraRedispatch: 1, + }); + }); + + test('every CASE_QUARANTINE entry is a diagnosed, dated, non-product blocking case within the tier cap', () => { + expect(quarantinePolicyProblems(CASE_QUARANTINE).map(problem => problem.message)).toEqual([]); + }); + + test('a quarantined case runs as isolated trial shards: its files register it literally', () => { + for (const id of Object.keys(CASE_QUARANTINE)) { + const files = (E2E_TOUCHFILES[id] ?? []).filter(file => /^test\/[^/]+\.test\.ts$/.test(file) && isPaidTestFile(file)); + expect(files.length, `${id}: no paid file registers it`).toBeGreaterThan(0); + for (const file of files) { + expect(fileCaseRegistration(file, fs.readFileSync(path.join(ROOT, file), 'utf8')).known, `${id}: ${file} registration must be statically known`).toBe(true); + } + } + }); +});