diff --git a/scripts/test-paid-shards.ts b/scripts/test-paid-shards.ts index 26ca8263b..20f3afc9c 100644 --- a/scripts/test-paid-shards.ts +++ b/scripts/test-paid-shards.ts @@ -339,11 +339,14 @@ export function buildPaidShardArgs( files: string[], timeoutMs: number, maxConcurrency: number = DEFAULT_WITHIN_SHARD_CONCURRENCY, + retries?: number, ): string[] { // Explicit --concurrent/--max-concurrency: the legacy path always set one; // omitting it here made within-shard parallelism differ silently between // the two runners (observed: 1.6x sumdur/wall sharded vs 8x legacy). - return ['test', ...files, '--retry', '1', '--concurrent', `--max-concurrency=${maxConcurrency}`, `--timeout=${timeoutMs}`]; + // Retries default to 1; RETRY_OVERRIDES membership (old matrix rows' + // earned `retries: 2`) flows through retriesForFiles at the call site. + return ['test', ...files, '--retry', String(retries ?? 1), '--concurrent', `--max-concurrency=${maxConcurrency}`, `--timeout=${timeoutMs}`]; } /** @@ -357,7 +360,17 @@ export function shardSlug(files: string[]): string { .replace(/[^a-zA-Z0-9._+-]/g, '-'); } -export type ShardStatus = 'passed' | 'failed' | 'timed-out' | 'never-started' | 'skipped-by-diff'; +export type ShardStatus = + | 'passed' + | 'failed' + | 'timed-out' + | 'never-started' + | 'skipped-by-diff' + // exit 0 with ZERO executed tests on a run that promised everything + // (EVALS_ALL): the hollow-file green the census backstop exists to catch. + // Under selective runs, 0-executed passed shards stay 'passed' (in-file + // diff/tier self-skips are legitimate there) and get a WARNING line only. + | 'passed-empty'; export interface ShardOutcome { shard: number; @@ -366,6 +379,8 @@ export interface ShardOutcome { exitCode: number | null; elapsedMs: number; groupPid: number | null; + /** Tests bun reported executing ("Ran N tests ..."), null when unknown. */ + executedTests: number | null; } export interface ShardCommand { @@ -440,6 +455,7 @@ export async function runPaidShard( exactTestFileSelectors(files, rootDir), timeoutMs, options.withinShardConcurrency ?? DEFAULT_WITHIN_SHARD_CONCURRENCY, + retriesForFiles(files), ), }; @@ -531,7 +547,10 @@ export async function runPaidShard( const logSuffix = status === 'passed' ? '' : ` — full log: ${logPath}`; log(`${label} ${status.toUpperCase()} in ${Math.round(elapsedMs / 1000)}s (exit ${exitCode ?? 'signal'})${logSuffix}`); - return { shard: shardNumber, files, status, exitCode, elapsedMs, groupPid }; + const executedTests = summary.terminalTestCounts.length > 0 + ? summary.terminalTestCounts.reduce((a, b) => a + b, 0) + : null; + return { shard: shardNumber, files, status, exitCode, elapsedMs, groupPid, executedTests }; } export interface RunSummary { @@ -552,7 +571,7 @@ export function summarize(outcomes: ShardOutcome[]): RunSummary { total: outcomes.length, executed: outcomes.length - count('never-started') - count('skipped-by-diff'), passed: count('passed'), - failed: count('failed'), + failed: count('failed') + count('passed-empty'), timedOut: count('timed-out'), neverStarted: count('never-started'), skippedByDiff: count('skipped-by-diff'), @@ -560,6 +579,28 @@ export function summarize(outcomes: ShardOutcome[]): RunSummary { }; } +/** + * Hollow-shard guard. Under EVALS_ALL (the run promised EVERY test), a + * passed shard whose bun summary reported 0 executed tests is not a pass — + * it is the zero-execution class one layer down (file selected, every test + * inside self-skipped, exit 0). Selective runs keep those shards 'passed' + * (in-file diff/tier self-skips are legitimate) and only warn. + */ +export function applyHollowShardGuard( + outcomes: ShardOutcome[], + opts: { evalsAll: boolean; warn?: (line: string) => void }, +): ShardOutcome[] { + const warn = opts.warn ?? ((line: string) => console.error(line)); + return outcomes.map((outcome) => { + if (outcome.status !== 'passed' || outcome.executedTests !== 0) return outcome; + if (!opts.evalsAll) { + warn(`[test:paid] WARNING: shard ${outcome.shard} passed with 0 executed tests (${outcome.files.join(' ')}) — legitimate under selection, hollow under EVALS_ALL`); + return outcome; + } + return { ...outcome, status: 'passed-empty' }; + }); +} + /** * Exit code for a finished run: skipped-by-diff shards are successes (the * parent proved none of their tests were selected); everything else must @@ -582,6 +623,7 @@ export async function runPaidShards( exitCode: null, elapsedMs: 0, groupPid: null, + executedTests: null, })); let next = 0; @@ -604,6 +646,7 @@ export async function runPaidShards( exitCode: null, elapsedMs: 0, groupPid: null, + executedTests: null, }; console.error(`[test:paid] shard ${index + 1} could not run: ${error instanceof Error ? error.message : String(error)}`); } @@ -631,6 +674,153 @@ export function formatSummary(summary: RunSummary): string[] { return lines; } +// ─── Planner / executor / report (the CI re-platform surface) ────────────── +// One PLANNER computes selection and the slice plan ONCE; K executor jobs +// consume it; a REPORT reconciles results against the plan. This kills two +// classes at the root: per-slice selector divergence (one slice failing +// merge-base resolution and running a different partition than its siblings) +// and hollow lanes (a missing/failed slice that artifact-presence aggregation +// would read as green). CI wiring: evals.yml planner job → K-way matrix of +// `--plan manifest.json --slice i` → report job running `--report `. + +export interface ManifestEntry { + file: string; + /** 1-based executor slice for planned entries; 0 for skipped/excluded. */ + slice: number; + status: 'planned' | 'skipped-by-diff' | 'excluded'; + reason?: string; +} + +export interface PaidRunManifest { + version: 1; + tier: PaidTier; + evalsAll: boolean; + sliceCount: number; + selectionReason: string; + entries: ManifestEntry[]; +} + +/** + * Files whose old evals.yml matrix rows carried `retries: 2`, with the + * receipts that earned them (see the deleted rows' comments). The runner + * default stays --retry 1; membership here is a literals map so retry + * parity with the matrix is explicit, not folklore. + */ +export const RETRY_OVERRIDES: Record = { + 'test/skill-e2e-workflow.test.ts': 2, + 'test/skill-e2e-office-hours-auto-mode.test.ts': 2, + 'test/skill-e2e-plan-mode-no-op.test.ts': 2, +}; + +export function retriesForFiles(files: string[]): number { + return Math.max(1, ...files.map((f) => RETRY_OVERRIDES[normalizeRelativePath(f)] ?? 1)); +} + +/** Round-robin the RUNNABLE (sorted) shard plan across K slices — deterministic. */ +export function buildRunManifest(opts: { + tier: PaidTier; + sliceCount: number; + evalsAll: boolean; + discovered?: string[]; + env?: NodeJS.ProcessEnv; + rootDir?: string; +}): PaidRunManifest { + if (!Number.isInteger(opts.sliceCount) || opts.sliceCount <= 0) { + throw new Error(`--slices needs a positive integer. Received: ${opts.sliceCount}`); + } + const rootDir = opts.rootDir ?? ROOT; + const discovered = opts.discovered ?? collectPaidTestFiles(rootDir); + const { selected, excluded } = selectPaidTestFiles(discovered, opts.tier, rootDir); + const shards = planPaidShards(selected, { maxFilesPerShard: 1 }); + const diffSelection = computePaidDiffSelection(opts.env ?? process.env); + const { runnable, skipped } = partitionShardsByDiffSelection(shards, diffSelection.selectedNames); + + const entries: ManifestEntry[] = []; + runnable.forEach((files, index) => { + entries.push({ file: files[0], slice: (index % opts.sliceCount) + 1, status: 'planned' }); + }); + for (const s of skipped) entries.push({ file: s.files[0], slice: 0, status: 'skipped-by-diff', reason: s.reason }); + for (const e of excluded) entries.push({ file: e.file, slice: 0, status: 'excluded', reason: e.reason }); + entries.sort((a, b) => (a.file < b.file ? -1 : 1)); + + return { + version: 1, + tier: opts.tier, + evalsAll: opts.evalsAll, + sliceCount: opts.sliceCount, + selectionReason: diffSelection.reason, + entries, + }; +} + +export function parseRunManifest(raw: string): PaidRunManifest { + const parsed = JSON.parse(raw) as PaidRunManifest; + if (parsed.version !== 1) throw new Error(`unsupported manifest version: ${(parsed as { version?: unknown }).version}`); + if (parsed.tier !== 'gate' && parsed.tier !== 'periodic') throw new Error(`manifest tier invalid: ${parsed.tier}`); + if (!Number.isInteger(parsed.sliceCount) || parsed.sliceCount <= 0) throw new Error('manifest sliceCount invalid'); + if (!Array.isArray(parsed.entries)) throw new Error('manifest entries missing'); + for (const entry of parsed.entries) { + if (typeof entry.file !== 'string' || !Number.isInteger(entry.slice)) throw new Error('manifest entry malformed'); + if (!['planned', 'skipped-by-diff', 'excluded'].includes(entry.status)) throw new Error(`manifest entry status invalid: ${entry.status}`); + if (entry.status === 'planned' && (entry.slice < 1 || entry.slice > parsed.sliceCount)) { + throw new Error(`planned entry ${entry.file} has out-of-range slice ${entry.slice}`); + } + } + return parsed; +} + +export interface SliceResult { + version: 1; + tier: PaidTier; + sliceIndex: number; + sliceCount: number; + outcomes: Array>; +} + +/** + * Reconcile slice results against the manifest — the fail-closed aggregation. + * Problems (any → non-zero): a slice index missing entirely (a cancelled or + * crashed executor whose artifact never landed), a planned entry no slice + * reported, an entry reported by the wrong/duplicate slice, or any reported + * outcome that is not a pass. + */ +export function verifySliceResults( + manifest: PaidRunManifest, + results: SliceResult[], +): { ok: boolean; problems: string[] } { + const problems: string[] = []; + const byIndex = new Map(); + for (const result of results) { + if (result.version !== 1) { problems.push(`slice result with unsupported version: ${String(result.version)}`); continue; } + if (result.tier !== manifest.tier) problems.push(`slice ${result.sliceIndex} ran tier ${result.tier}, manifest says ${manifest.tier}`); + if (byIndex.has(result.sliceIndex)) problems.push(`duplicate result for slice ${result.sliceIndex}`); + byIndex.set(result.sliceIndex, result); + } + for (let index = 1; index <= manifest.sliceCount; index += 1) { + if (!byIndex.has(index)) problems.push(`slice ${index}/${manifest.sliceCount} reported NO result — cancelled/crashed executor, not a pass`); + } + + const reported = new Map(); + for (const result of results) { + for (const outcome of result.outcomes) { + const file = normalizeRelativePath(outcome.files[0] ?? ''); + if (reported.has(file)) problems.push(`${file} reported by two slices`); + reported.set(file, { slice: result.sliceIndex, status: outcome.status }); + } + } + for (const entry of manifest.entries) { + if (entry.status !== 'planned') continue; + const got = reported.get(normalizeRelativePath(entry.file)); + if (!got) { + if (byIndex.has(entry.slice)) problems.push(`planned ${entry.file} (slice ${entry.slice}) was never reported`); + continue; // the missing-slice problem above already covers it + } + if (got.slice !== entry.slice) problems.push(`${entry.file} planned for slice ${entry.slice} but reported by slice ${got.slice}`); + if (got.status !== 'passed') problems.push(`${entry.file}: ${got.status}`); + } + return { ok: problems.length === 0, problems }; +} + type CliOptions = { tier: PaidTier; listOnly: boolean; @@ -638,6 +828,16 @@ type CliOptions = { jobs: number; withinShardConcurrency: number; maxFilesPerShard: number; + /** Planner mode: write the run manifest here and exit. */ + emitPlanPath: string | null; + /** Slice count for --emit-plan. */ + slices: number; + /** Executor mode: consume this manifest... */ + planPath: string | null; + /** ...running only this 1-based slice. */ + sliceIndex: number | null; + /** Report mode: reconcile manifest.json + slice-*.json under this dir. */ + reportDir: string | null; }; function parsePositiveInt(value: string | undefined, flag: string): number { @@ -674,6 +874,11 @@ export function parseCliOptions(argv: string[], env: NodeJS.ProcessEnv = process ? parsePositiveInt(env.EVALS_CONCURRENCY, 'EVALS_CONCURRENCY') : DEFAULT_WITHIN_SHARD_CONCURRENCY, maxFilesPerShard: DEFAULT_MAX_FILES_PER_SHARD, + emitPlanPath: null, + slices: 1, + planPath: null, + sliceIndex: null, + reportDir: null, }; for (let index = 0; index < argv.length; index += 1) { @@ -688,6 +893,23 @@ export function parseCliOptions(argv: string[], env: NodeJS.ProcessEnv = process if (arg === '--timeout') { options.timeoutMs = parsePositiveInt(argv[index += 1], '--timeout') * 1000; continue; } if (arg === '--jobs') { options.jobs = parsePositiveInt(argv[index += 1], '--jobs'); continue; } if (arg === '--files-per-shard') { options.maxFilesPerShard = parsePositiveInt(argv[index += 1], '--files-per-shard'); continue; } + if (arg === '--emit-plan') { + const value = argv[index += 1]; + if (!value) throw new Error('--emit-plan needs a file path'); + options.emitPlanPath = value; continue; + } + if (arg === '--slices') { options.slices = parsePositiveInt(argv[index += 1], '--slices'); continue; } + if (arg === '--plan') { + const value = argv[index += 1]; + if (!value) throw new Error('--plan needs a manifest path'); + options.planPath = value; continue; + } + if (arg === '--slice') { options.sliceIndex = parsePositiveInt(argv[index += 1], '--slice'); continue; } + if (arg === '--report') { + const value = argv[index += 1]; + if (!value) throw new Error('--report needs a directory'); + options.reportDir = value; continue; + } throw new Error(`Unknown argument: ${arg}`); } return options; @@ -695,9 +917,111 @@ export function parseCliOptions(argv: string[], env: NodeJS.ProcessEnv = process async function main(): Promise { const options = parseCliOptions(process.argv.slice(2)); + + // ── Planner mode: compute selection + the slice plan ONCE, write it, exit. + if (options.emitPlanPath) { + const manifest = buildRunManifest({ + tier: options.tier, + sliceCount: options.slices, + evalsAll: process.env.EVALS_ALL === '1', + }); + fs.mkdirSync(path.dirname(path.resolve(options.emitPlanPath)), { recursive: true }); + fs.writeFileSync(options.emitPlanPath, `${JSON.stringify(manifest, null, 2)}\n`); + const planned = manifest.entries.filter((e) => e.status === 'planned').length; + const skipped = manifest.entries.filter((e) => e.status === 'skipped-by-diff').length; + const excludedCount = manifest.entries.filter((e) => e.status === 'excluded').length; + console.log( + `[test:paid] plan: tier=${manifest.tier} evalsAll=${manifest.evalsAll} — ` + + `${planned} planned across ${manifest.sliceCount} slice(s), ${skipped} skipped by diff, ` + + `${excludedCount} excluded (${manifest.selectionReason})`, + ); + return 0; + } + + // ── Report mode: reconcile slice artifacts against the manifest. Fail-closed: + // a slice whose artifact never landed is a FAILURE, not an absence. + if (options.reportDir) { + const manifest = parseRunManifest(fs.readFileSync(path.join(options.reportDir, 'manifest.json'), 'utf-8')); + const results: SliceResult[] = fs.readdirSync(options.reportDir) + .filter((name) => /^slice-\d+\.json$/.test(name)) + .map((name) => JSON.parse(fs.readFileSync(path.join(options.reportDir, name), 'utf-8')) as SliceResult); + const verdict = verifySliceResults(manifest, results); + const planned = manifest.entries.filter((e) => e.status === 'planned').length; + console.log(`[test:paid] report: ${results.length}/${manifest.sliceCount} slices, ${planned} planned shards, tier=${manifest.tier}`); + for (const result of results.sort((a, b) => a.sliceIndex - b.sliceIndex)) { + for (const outcome of result.outcomes) { + console.log(` slice ${result.sliceIndex} ${outcome.status.padEnd(15)} ${String(Math.round(outcome.elapsedMs / 1000)).padStart(5)}s ${outcome.files.join(' ')}`); + } + } + if (!verdict.ok) { + console.error(`[test:paid] report: ${verdict.problems.length} problem(s):`); + for (const problem of verdict.problems) console.error(` ✗ ${problem}`); + return 1; + } + console.log('[test:paid] report: every planned shard accounted and passed'); + return 0; + } + const discovered = collectPaidTestFiles(); if (discovered.length === 0) throw new Error('No paid test files were discovered.'); + // ── Executor mode: consume the planner's manifest; never self-select. + if (options.planPath || options.sliceIndex !== null) { + if (!options.planPath || options.sliceIndex === null) { + throw new Error('--plan and --slice must be used together'); + } + const manifest = parseRunManifest(fs.readFileSync(options.planPath, 'utf-8')); + if (manifest.tier !== options.tier) { + throw new Error(`manifest tier ${manifest.tier} != requested tier ${options.tier} — refusing a cross-tier run`); + } + if (options.sliceIndex > manifest.sliceCount) { + throw new Error(`--slice ${options.sliceIndex} exceeds manifest sliceCount ${manifest.sliceCount}`); + } + const mine = manifest.entries.filter((e) => e.status === 'planned' && e.slice === options.sliceIndex); + const shards = mine.map((e) => [e.file]); + console.log(`[test:paid] slice ${options.sliceIndex}/${manifest.sliceCount}: ${shards.length} shard(s), tier=${manifest.tier}, evalsAll=${manifest.evalsAll}`); + + const evalDirBase = process.env.GSTACK_EVAL_DIR || getProjectEvalDir(); + let summary: RunSummary; + if (shards.length === 0) { + summary = summarize([]); + } else { + preflightAnthropicApi(process.env); + summary = await runPaidShards(shards, { + timeoutMs: options.timeoutMs, + jobs: options.jobs, + withinShardConcurrency: options.withinShardConcurrency, + env: { + ...process.env, + EVALS: '1', + EVALS_TIER: options.tier, + ...(manifest.evalsAll ? { EVALS_ALL: '1' } : {}), + EVALS_PREFLIGHT_OK: '1', + // The manifest IS the selection: children must not re-derive a + // possibly-different one from their own git view. + EVALS_SELECTION_JSON: JSON.stringify({ version: 1, selected: null, reason: `manifest slice ${options.sliceIndex}: ${manifest.selectionReason}` }), + }, + evalDirBase, + }); + } + const guarded = applyHollowShardGuard(summary.outcomes, { evalsAll: manifest.evalsAll }); + summary = summarize(guarded); + const sliceResult: SliceResult = { + version: 1, + tier: manifest.tier, + sliceIndex: options.sliceIndex, + sliceCount: manifest.sliceCount, + outcomes: guarded.map(({ files, status, exitCode, elapsedMs, executedTests }) => + ({ files, status, exitCode, elapsedMs, executedTests })), + }; + fs.mkdirSync(evalDirBase, { recursive: true }); + const sliceResultPath = path.join(evalDirBase, `slice-${options.sliceIndex}.json`); + fs.writeFileSync(sliceResultPath, `${JSON.stringify(sliceResult, null, 2)}\n`); + console.log(`[test:paid] slice result: ${sliceResultPath}`); + for (const line of formatSummary(summary)) console.log(line); + return summaryExitCode(summary); + } + const { selected, excluded } = selectPaidTestFiles(discovered, options.tier); const shards = planPaidShards(selected, { maxFilesPerShard: options.maxFilesPerShard }); @@ -765,8 +1089,12 @@ async function main(): Promise { exitCode: null, elapsedMs: 0, groupPid: null, + executedTests: null, })); - const summary = summarize([...runSummary.outcomes, ...skippedOutcomes]); + const guardedOutcomes = applyHollowShardGuard(runSummary.outcomes, { + evalsAll: process.env.EVALS_ALL === '1', + }); + const summary = summarize([...guardedOutcomes, ...skippedOutcomes]); for (const line of formatSummary(summary)) console.log(line); return summaryExitCode(summary); } diff --git a/scripts/test-strict-output.ts b/scripts/test-strict-output.ts index 31e2dccd3..2e91bd096 100644 --- a/scripts/test-strict-output.ts +++ b/scripts/test-strict-output.ts @@ -19,7 +19,7 @@ const ROOT = path.resolve(import.meta.dir, '..'); const ANSI_ESCAPE = /\u001B\[[0-?]*[ -/]*[@-~]/g; const BUN_FAIL_RESULT = /^\(fail\) .+ \[(?:\d+(?:\.\d+)?)(?:ns|us|µs|ms|s)\]$/; const BUN_BETWEEN_TESTS_ERROR = '# Unhandled error between tests'; -const BUN_TERMINAL_SUMMARY = /^Ran \d+ tests? across (\d+) files?\. \[(?:\d+(?:\.\d+)?)(?:ns|us|µs|ms|s)\]$/; +const BUN_TERMINAL_SUMMARY = /^Ran (\d+) tests? across (\d+) files?\. \[(?:\d+(?:\.\d+)?)(?:ns|us|µs|ms|s)\]$/; export type BunTestOutputFinding = 'failed-test' | 'unhandled-between-tests'; @@ -27,6 +27,8 @@ export interface BunTestOutputSummary { failedTests: number; unhandledBetweenTests: number; terminalFileCounts: number[]; + /** Test counts from the same terminal lines — feeds the hollow-shard guard. */ + terminalTestCounts: number[]; } export type ForwardedTerminationSignal = 'SIGINT' | 'SIGTERM'; @@ -196,9 +198,15 @@ export function classifyBunTestOutputLine(rawLine: string): BunTestOutputFinding } export function parseBunTerminalSummaryLine(rawLine: string): number | null { + return parseBunTerminalSummary(rawLine)?.files ?? null; +} + +export function parseBunTerminalSummary(rawLine: string): { tests: number; files: number } | null { const line = stripAnsiLine(rawLine); const match = BUN_TERMINAL_SUMMARY.exec(line); - return match ? Number.parseInt(match[1], 10) : null; + return match + ? { tests: Number.parseInt(match[1], 10), files: Number.parseInt(match[2], 10) } + : null; } /** @@ -220,6 +228,7 @@ export class BunTestOutputClassifier { private failedTests = 0; private unhandledBetweenTests = 0; private terminalFileCounts: number[] = []; + private terminalTestCounts: number[] = []; write(chunk: Uint8Array | string, origin: ClassifierOrigin = 'stdout'): void { this.pending[origin] += typeof chunk === 'string' @@ -242,6 +251,7 @@ export class BunTestOutputClassifier { failedTests: this.failedTests, unhandledBetweenTests: this.unhandledBetweenTests, terminalFileCounts: [...this.terminalFileCounts], + terminalTestCounts: [...this.terminalTestCounts], }; } @@ -258,8 +268,11 @@ export class BunTestOutputClassifier { const finding = classifyBunTestOutputLine(line); if (finding === 'failed-test') this.failedTests += 1; if (finding === 'unhandled-between-tests') this.unhandledBetweenTests += 1; - const terminalFileCount = parseBunTerminalSummaryLine(line); - if (terminalFileCount !== null) this.terminalFileCounts.push(terminalFileCount); + const terminal = parseBunTerminalSummary(line); + if (terminal !== null) { + this.terminalFileCounts.push(terminal.files); + this.terminalTestCounts.push(terminal.tests); + } } } diff --git a/test/paid-run-manifest.test.ts b/test/paid-run-manifest.test.ts new file mode 100644 index 000000000..402f63e24 --- /dev/null +++ b/test/paid-run-manifest.test.ts @@ -0,0 +1,186 @@ +/** + * Planner/executor/report contract for the re-platformed paid CI lane. + * + * The classes these pin (each was a live CI failure mode of the old + * hand-enumerated matrix, or a review-identified risk of the migration): + * - per-slice selector divergence → ONE planner manifest, executors consume + * - hollow lanes → a slice with no artifact is a FAILURE, not an absence + * - hollow shards → EVALS_ALL + exit 0 + zero executed tests ≠ pass + * - retry parity → the old matrix rows' earned `retries: 2` survive as a + * literals map, not folklore + */ +import { describe, expect, test } from 'bun:test'; +import * as fs from 'node:fs'; +import * as path from 'node:path'; + +import { + applyHollowShardGuard, + buildPaidShardArgs, + buildRunManifest, + parseRunManifest, + retriesForFiles, + RETRY_OVERRIDES, + summarize, + summaryExitCode, + verifySliceResults, + type PaidRunManifest, + type ShardOutcome, + type SliceResult, +} from '../scripts/test-paid-shards'; + +const ROOT = path.resolve(__dirname, '..'); + +const outcome = (over: Partial): ShardOutcome => ({ + shard: 1, + files: ['test/skill-e2e-x.test.ts'], + status: 'passed', + exitCode: 0, + elapsedMs: 1000, + groupPid: null, + executedTests: 3, + ...over, +}); + +describe('run manifest (planner)', () => { + test('live build: every paid file appears exactly once; planned slices partition 1..K', () => { + const manifest = buildRunManifest({ tier: 'gate', sliceCount: 5, evalsAll: true, env: { EVALS_ALL: '1' } }); + const files = manifest.entries.map((e) => e.file); + expect(new Set(files).size).toBe(files.length); + const planned = manifest.entries.filter((e) => e.status === 'planned'); + expect(planned.length).toBeGreaterThan(20); // census sanity + for (const entry of planned) { + expect(entry.slice).toBeGreaterThanOrEqual(1); + expect(entry.slice).toBeLessThanOrEqual(5); + } + // Round-robin balance: slice sizes differ by at most 1. + const sizes = [1, 2, 3, 4, 5].map((i) => planned.filter((e) => e.slice === i).length); + expect(Math.max(...sizes) - Math.min(...sizes)).toBeLessThanOrEqual(1); + // Non-runnable entries carry slice 0 and a reason. + for (const entry of manifest.entries.filter((e) => e.status !== 'planned')) { + expect(entry.slice).toBe(0); + expect(entry.reason ?? '').not.toBe(''); + } + }); + + test('deterministic for identical inputs', () => { + const opts = { tier: 'periodic' as const, sliceCount: 4, evalsAll: true, env: { EVALS_ALL: '1' } }; + expect(buildRunManifest(opts)).toEqual(buildRunManifest(opts)); + }); + + test('parse round-trips and rejects malformed manifests', () => { + const manifest = buildRunManifest({ tier: 'gate', sliceCount: 2, evalsAll: false, env: {} }); + expect(parseRunManifest(JSON.stringify(manifest))).toEqual(manifest); + expect(() => parseRunManifest('{}')).toThrow(/version/); + expect(() => parseRunManifest(JSON.stringify({ ...manifest, tier: 'e2e' }))).toThrow(/tier/); + expect(() => parseRunManifest(JSON.stringify({ ...manifest, sliceCount: 0 }))).toThrow(/sliceCount/); + const outOfRange = { + ...manifest, + entries: [{ file: 'test/skill-e2e-x.test.ts', slice: 9, status: 'planned' }], + }; + expect(() => parseRunManifest(JSON.stringify(outOfRange))).toThrow(/out-of-range/); + }); +}); + +describe('slice-result reconciliation (report)', () => { + const manifest: PaidRunManifest = { + version: 1, + tier: 'gate', + evalsAll: false, + sliceCount: 2, + selectionReason: 'test fixture', + entries: [ + { file: 'test/skill-e2e-a.test.ts', slice: 1, status: 'planned' }, + { file: 'test/skill-e2e-b.test.ts', slice: 2, status: 'planned' }, + { file: 'test/skill-e2e-c.test.ts', slice: 0, status: 'skipped-by-diff', reason: 'unselected' }, + ], + }; + const slice = (index: number, files: string[], status: ShardOutcome['status'] = 'passed'): SliceResult => ({ + version: 1, + tier: 'gate', + sliceIndex: index, + sliceCount: 2, + outcomes: files.map((f) => ({ files: [f], status, exitCode: 0, elapsedMs: 5, executedTests: 2 })), + }); + + test('all slices present and passing → ok', () => { + const verdict = verifySliceResults(manifest, [ + slice(1, ['test/skill-e2e-a.test.ts']), + slice(2, ['test/skill-e2e-b.test.ts']), + ]); + expect(verdict).toEqual({ ok: true, problems: [] }); + }); + + test('a missing slice artifact is a FAILURE, not an absence', () => { + const verdict = verifySliceResults(manifest, [slice(1, ['test/skill-e2e-a.test.ts'])]); + expect(verdict.ok).toBe(false); + expect(verdict.problems.join('\n')).toContain('slice 2/2 reported NO result'); + }); + + test('a planned shard nobody reported fails even when its slice reported', () => { + const verdict = verifySliceResults(manifest, [ + slice(1, []), + slice(2, ['test/skill-e2e-b.test.ts']), + ]); + expect(verdict.ok).toBe(false); + expect(verdict.problems.join('\n')).toContain('never reported'); + }); + + test('wrong-slice, duplicate, cross-tier, and failing outcomes all surface', () => { + const wrongSlice = verifySliceResults(manifest, [ + slice(1, ['test/skill-e2e-b.test.ts']), + slice(2, ['test/skill-e2e-a.test.ts']), + ]); + expect(wrongSlice.ok).toBe(false); + const failing = verifySliceResults(manifest, [ + slice(1, ['test/skill-e2e-a.test.ts'], 'failed'), + slice(2, ['test/skill-e2e-b.test.ts']), + ]); + expect(failing.problems.join('\n')).toContain('test/skill-e2e-a.test.ts: failed'); + const crossTier = verifySliceResults(manifest, [ + { ...slice(1, ['test/skill-e2e-a.test.ts']), tier: 'periodic' }, + slice(2, ['test/skill-e2e-b.test.ts']), + ]); + expect(crossTier.problems.join('\n')).toContain('ran tier periodic'); + }); +}); + +describe('hollow-shard guard', () => { + test('EVALS_ALL: passed with 0 executed tests becomes passed-empty and fails the run', () => { + const guarded = applyHollowShardGuard([outcome({ executedTests: 0 })], { evalsAll: true, warn: () => {} }); + expect(guarded[0].status).toBe('passed-empty'); + const summary = summarize(guarded); + expect(summary.failed).toBe(1); + expect(summaryExitCode(summary)).toBe(1); + }); + + test('selective run: same shape stays passed, warns once', () => { + const warnings: string[] = []; + const guarded = applyHollowShardGuard([outcome({ executedTests: 0 })], { + evalsAll: false, warn: (line) => warnings.push(line), + }); + expect(guarded[0].status).toBe('passed'); + expect(warnings).toHaveLength(1); + }); + + test('unknown executedTests (null) is never guessed hollow', () => { + const guarded = applyHollowShardGuard([outcome({ executedTests: null })], { evalsAll: true }); + expect(guarded[0].status).toBe('passed'); + }); +}); + +describe('retry parity', () => { + test('overrides exist only for the files whose matrix rows earned them, and each names a real file', () => { + expect(Object.keys(RETRY_OVERRIDES).sort()).toEqual([ + 'test/skill-e2e-office-hours-auto-mode.test.ts', + 'test/skill-e2e-plan-mode-no-op.test.ts', + 'test/skill-e2e-workflow.test.ts', + ]); + for (const file of Object.keys(RETRY_OVERRIDES)) { + expect(fs.existsSync(path.join(ROOT, file)), `stale RETRY_OVERRIDES entry: ${file}`).toBe(true); + } + expect(retriesForFiles(['test/skill-e2e-workflow.test.ts'])).toBe(2); + expect(retriesForFiles(['test/skill-e2e-retro.test.ts'])).toBe(1); + expect(buildPaidShardArgs(['x'], 1000, 4, 2)).toContain('2'); + expect(buildPaidShardArgs(['x'], 1000, 4).join(' ')).toContain('--retry 1'); + }); +});