mirror of
https://github.com/garrytan/gstack.git
synced 2026-10-03 01:46:55 +02:00
feat(evals): ~12-minute blocking paid lanes and a non-blocking marathon lane
- Planner budget mode (--slice-budget S --jobs J): recorded per-tier wall times pack into as many ~9-minute executors as the work needs; the plan records per-slice estimates and the CI job timeout (supervised worst case + 20 min). evals.yml and evals-periodic.yml derive matrix size and timeout-minutes from it; max-parallel covers every slice at once. - Case shards: plan/design/review-army/shared-libs(-paths) run one registered case per process (<file>#<case id>, exact name pattern, exactly one case). - Retry rule: a timed-out attempt is a verdict. Only files whose every case budget is CAPTURE tier or shorter keep one retry; walls shrink to match. - Marathon tier: positive selection, excluded from gate/periodic planners, run by the new evals-marathon.yml (weekly + dispatch, fresh, own report). - PR-lane E2E reuse of verified first-attempt passes on identical inputs; the report rejects reuse outside the fast PR profile. - Duration seed from census run 36385945043, per tier and per case shard.
This commit is contained in:
1 parent
acb7bc02b1
commit
0023d011a3
18 files changed
+1773
-450
No files matched your search
@@ -5,7 +5,7 @@ import * as path from 'node:path';
|
||||
import { runPaidShard, shardSlug } from '../scripts/test-paid-shards';
|
||||
|
||||
const ROOT = path.resolve(import.meta.dir, '..');
|
||||
const workflows = ['evals.yml', 'evals-periodic.yml'].map(name => ({
|
||||
const workflows = ['evals.yml', 'evals-periodic.yml', 'evals-marathon.yml'].map(name => ({
|
||||
name,
|
||||
value: Bun.YAML.parse(fs.readFileSync(path.join(ROOT, '.github/workflows', name), 'utf8')) as any,
|
||||
}));
|
||||
@@ -23,7 +23,7 @@ function render(template: string, fields: Record<string, string>): string {
|
||||
|
||||
test('every direct CI paid executor binds a safe unique run/attempt/job/slice identity', () => {
|
||||
expect(executors.map(({ name, jobName }) => `${name}:${jobName}`)).toEqual([
|
||||
'evals.yml:eval-slices', 'evals-periodic.yml:eval-slices', 'evals-periodic.yml:gate-census',
|
||||
'evals.yml:eval-slices', 'evals-periodic.yml:eval-slices', 'evals-periodic.yml:gate-census', 'evals-marathon.yml:eval-slices',
|
||||
]);
|
||||
const ids = new Set<string>();
|
||||
for (const [workflowIndex, { job, step }] of executors.entries()) {
|
||||
@@ -32,9 +32,11 @@ test('every direct CI paid executor binds a safe unique run/attempt/job/slice id
|
||||
expect(env.EVALS_RUN_ID).toBeString();
|
||||
for (const run of ['36302678692', '36302678693']) {
|
||||
for (const attempt of ['1', '2']) {
|
||||
for (const slice of job.strategy.matrix.slice) {
|
||||
// The planner sizes the matrix; cover more slices than any live plan.
|
||||
expect(job.strategy.matrix.slice).toMatch(/^\$\{\{ fromJSON\(needs\.plan-slices\.outputs\.(?:[a-z]+_)?slices\) \}\}$/);
|
||||
for (let slice = 1; slice <= 64; slice++) {
|
||||
const id = render(env.EVALS_RUN_ID, {
|
||||
'github.run_id': `${run}${workflowIndex === 0 ? '0' : '1'}`,
|
||||
'github.run_id': `${run}${workflowIndex}`,
|
||||
'github.run_attempt': attempt, 'matrix.slice': String(slice),
|
||||
});
|
||||
expect(id).toMatch(/^[A-Za-z0-9_-]+$/);
|
||||
|
||||
@@ -30,8 +30,10 @@ test('the actual CI cookie repair planner executes only eight dependent cases wi
|
||||
expect(manifest.evalsAll).toBe(false);
|
||||
expect(manifest.selection).toEqual({ e2e: ['browse-basic', 'browse-snapshot', 'qa-quick', 'qa-only-no-fix', 'design-review-detector-shim-dom', 'diagram-triplet', 'canary-workflow', 'benchmark-workflow'], judges: [] });
|
||||
expect(manifest.entries.filter(entry => entry.status === 'planned').map(entry => entry.file).sort()).toEqual([
|
||||
'test/skill-e2e-bws.test.ts', 'test/skill-e2e-deploy.test.ts', 'test/skill-e2e-design.test.ts', 'test/skill-e2e-diagram.test.ts', 'test/skill-e2e-qa-workflow.test.ts',
|
||||
'test/skill-e2e-bws.test.ts', 'test/skill-e2e-deploy.test.ts', 'test/skill-e2e-design.test.ts#design-review-detector-shim-dom', 'test/skill-e2e-diagram.test.ts', 'test/skill-e2e-qa-workflow.test.ts',
|
||||
]);
|
||||
// The case-sharded design file runs only its one selected cookie case.
|
||||
expect(manifest.entries.filter(entry => entry.file.startsWith('test/skill-e2e-design.test.ts#') && entry.status === 'skipped-by-diff').length).toBeGreaterThan(0);
|
||||
});
|
||||
|
||||
test('the existing quality and behavior phases retain their complete separate shard census', () => {
|
||||
@@ -42,7 +44,10 @@ test('the existing quality and behavior phases retain their complete separate sh
|
||||
expect(quality.evalsAll).toBe(true);
|
||||
expect(behavior.evalsAll).toBe(true);
|
||||
expect(qualityFiles).toHaveLength(1);
|
||||
expect(behaviorFiles).toHaveLength(45);
|
||||
// 44 files (first-task-scaffold registers no gate case, so the gate lane
|
||||
// skips it); the five case-sharded files contribute one shard per gate case.
|
||||
expect(new Set(behaviorFiles.map(file => file.split('#')[0])).size).toBe(44);
|
||||
expect(behaviorFiles).toHaveLength(62);
|
||||
expect(behaviorFiles).toEqual(expect.arrayContaining([
|
||||
'test/skill-e2e-qa-callers.test.ts',
|
||||
'test/skill-e2e-qa-functional-fix.test.ts',
|
||||
|
||||
@@ -0,0 +1,145 @@
|
||||
import { describe, expect, test } from 'bun:test';
|
||||
import * as fs from 'node:fs';
|
||||
import * as os from 'node:os';
|
||||
import * as path from 'node:path';
|
||||
import {
|
||||
e2eReuseEnvironment, e2eReuseLaneProblem, e2eShardIdentity, e2eShardInputFiles, prepareE2EShardReuse,
|
||||
type E2EShardReuseRequest,
|
||||
} from '../scripts/e2e-shard-reuse';
|
||||
import { buildRunManifest, fileCaseRegistration, runPaidShard, verifySliceResults, type SliceResult } from '../scripts/test-paid-shards';
|
||||
|
||||
const ROOT = path.resolve(import.meta.dir, '..');
|
||||
const FILE = 'test/skill-e2e-deploy.test.ts';
|
||||
const scratch = fs.mkdtempSync(path.join(os.tmpdir(), 'e2e-reuse-'));
|
||||
const bin = path.join(scratch, 'bin');
|
||||
fs.mkdirSync(bin);
|
||||
fs.writeFileSync(path.join(bin, 'claude'), '#!/bin/sh\necho "9.9.9 (Claude Code)"\n', { mode: 0o755 });
|
||||
|
||||
const laneEnv = (over: NodeJS.ProcessEnv = {}): NodeJS.ProcessEnv => ({
|
||||
PATH: `${bin}${path.delimiter}${process.env.PATH}`, HOME: scratch,
|
||||
EVALS_TIER: 'gate', EVALS_PROFILE: 'pr', EVALS: '1',
|
||||
EVALS_CACHE_DIR: path.join(scratch, 'cache'), EVALS_CACHE_REPOSITORY: 'garrytan/gstack', EVALS_CACHE_PR: '42',
|
||||
EVALS_CACHE_RUNTIME_ID: 'a'.repeat(64), GITHUB_RUN_ID: '1001', GITHUB_RUN_ATTEMPT: '1', ANTHROPIC_API_KEY: 'sk-fixture',
|
||||
GSTACK_CLAUDE_CLI_VERSION: '9.9.9 (Claude Code)',
|
||||
...over,
|
||||
});
|
||||
|
||||
function request(over: Partial<E2EShardReuseRequest> = {}): E2EShardReuseRequest {
|
||||
const { registered, known } = fileCaseRegistration(FILE, fs.readFileSync(path.join(ROOT, FILE), 'utf8'));
|
||||
return { root: ROOT, key: FILE, file: FILE, caseIds: ['setup-deploy-workflow'], registeredIds: registered, registrationKnown: known,
|
||||
casePattern: '(?:^|\\s)(?:setup-deploy-workflow)$', expectedCases: 1, retries: 0, timeoutMs: 1_800_000,
|
||||
withinShardConcurrency: 2, tier: 'gate', profile: 'pr', env: laneEnv(), ...over };
|
||||
}
|
||||
|
||||
describe('E2E shard reuse eligibility', () => {
|
||||
test('only the same-PR fast profile with an immutable runtime and default endpoint may reuse', () => {
|
||||
expect(e2eReuseLaneProblem(laneEnv(), 'pr')).toBeNull();
|
||||
for (const [env, mode, problem] of [
|
||||
[laneEnv(), 'full-fallback', 'Only the fast PR profile'],
|
||||
[laneEnv(), undefined, 'Only the fast PR profile'],
|
||||
[laneEnv({ EVALS_CACHE_PR: '' }), 'pr', 'same-PR cache scope'],
|
||||
[laneEnv({ EVALS_CACHE_RUNTIME_ID: 'latest' }), 'pr', 'immutable runtime'],
|
||||
[laneEnv({ EVALS_FRESH: '1' }), 'pr', 'Fresh validation'],
|
||||
[laneEnv({ EVALS_TIER: 'periodic' }), 'pr', 'Fresh validation'],
|
||||
[laneEnv({ EVALS_CACHE_PURPOSE: 'periodic' }), 'pr', 'execute fresh'],
|
||||
[laneEnv({ EVALS_CACHE_PURPOSE: 'marathon' }), 'pr', 'execute fresh'],
|
||||
[laneEnv({ EVALS_CACHE_PURPOSE: 'release' }), 'pr', 'execute fresh'],
|
||||
[laneEnv({ NODE_OPTIONS: '--require x' }), 'pr', 'Preload'],
|
||||
[laneEnv({ ANTHROPIC_BASE_URL: 'https://proxy.example' }), 'pr', 'Custom model endpoint'],
|
||||
] as const) expect(e2eReuseLaneProblem(env, mode)).toContain(problem);
|
||||
});
|
||||
|
||||
test('the identity binds the child environment except run-scoped transport, and never secret values', () => {
|
||||
const env = e2eReuseEnvironment(laneEnv({ EVALS_RUN_ID: 'run-1', GSTACK_EVAL_DIR: '/tmp/x', EVALS_SELECTION_JSON: '{}', EVALS_MODEL: 'm', UNRELATED: 'x' }));
|
||||
expect(env.ANTHROPIC_API_KEY).toBe('set');
|
||||
expect(env.EVALS_MODEL).toBe('m');
|
||||
for (const name of ['EVALS_RUN_ID', 'GSTACK_EVAL_DIR', 'EVALS_SELECTION_JSON', 'EVALS_CACHE_DIR', 'EVALS_CACHE_PR', 'UNRELATED', 'GITHUB_RUN_ID']) {
|
||||
expect(env[name], name).toBeUndefined();
|
||||
}
|
||||
expect(JSON.stringify(env)).not.toContain('sk-fixture');
|
||||
});
|
||||
|
||||
test('consumed files cover the test closure, every registered touchfile, the globals and the harness', () => {
|
||||
const files = e2eShardInputFiles(request());
|
||||
for (const file of [FILE, 'test/helpers/e2e-helpers.ts', 'scripts/test-paid-shards.ts', 'scripts/e2e-shard-reuse.ts',
|
||||
'bun.lock', '.github/workflows/evals.yml', '.github/actions/register-gstack-skills/action.yml', '.github/docker/Dockerfile.ci',
|
||||
'setup-deploy/SKILL.md.tmpl', 'test/helpers/touchfiles-data.ts']) expect(files, file).toContain(file);
|
||||
expect(files).not.toContain('package.json');
|
||||
expect(files.some(file => file.startsWith('node_modules/'))).toBe(true);
|
||||
expect(() => e2eShardInputFiles({ ...request(), registeredIds: ['no-such-case'] })).not.toThrow();
|
||||
});
|
||||
|
||||
test('unknown or unprovable inputs fail closed', () => {
|
||||
expect(e2eShardIdentity(request()).status).toBe('eligible');
|
||||
for (const [over, reason] of [
|
||||
[{ retries: 1 }, 'first attempt'],
|
||||
[{ registrationKnown: false }, 'statically complete'],
|
||||
[{ caseIds: [] }, 'exactly known'],
|
||||
[{ expectedCases: 2 }, 'exactly known'],
|
||||
[{ caseIds: ['not-registered'] }, 'exactly known'],
|
||||
[{ file: 'test/skill-llm-eval.test.ts', key: 'test/skill-llm-eval.test.ts' }, 'audited E2E file'],
|
||||
[{ env: laneEnv({ PATH: path.join(scratch, 'empty') }) }, 'Claude CLI version is unknown'],
|
||||
] as const) {
|
||||
const result = e2eShardIdentity(request(over as Partial<E2EShardReuseRequest>));
|
||||
expect(result.status, reason).toBe('ineligible');
|
||||
expect(result.status === 'ineligible' ? result.reason : '').toContain(reason);
|
||||
}
|
||||
});
|
||||
|
||||
test('any consumed parameter, pin or runtime change is a different identity', () => {
|
||||
const key = (over: Partial<E2EShardReuseRequest>) => {
|
||||
const result = e2eShardIdentity(request(over));
|
||||
if (result.status !== 'eligible') throw new Error(result.reason);
|
||||
return result.identity.key;
|
||||
};
|
||||
const base = key({});
|
||||
expect(key({})).toBe(base);
|
||||
expect(key({ env: laneEnv({ EVALS_RUN_ID: 'another-run', GITHUB_RUN_ID: '9' }) })).toBe(base);
|
||||
for (const over of [{ timeoutMs: 1_000 }, { withinShardConcurrency: 1 }, { casePattern: 'x' },
|
||||
{ env: laneEnv({ EVALS_MODEL: 'other' }) }, { env: laneEnv({ EVALS_CACHE_RUNTIME_ID: 'b'.repeat(64) }) },
|
||||
{ env: laneEnv({ EVALS_CACHE_PR: '43' }) }] as Array<Partial<E2EShardReuseRequest>>) expect(key(over)).not.toBe(base);
|
||||
});
|
||||
});
|
||||
|
||||
describe('E2E shard reuse through the runner', () => {
|
||||
test('a fresh first-attempt pass publishes; identical inputs then reuse without launching; changed inputs run', async () => {
|
||||
const env = laneEnv({ EVALS_CACHE_DIR: path.join(scratch, 'roundtrip') });
|
||||
const first = prepareE2EShardReuse(request({ env }))!;
|
||||
expect(first.lookup()).toBeNull();
|
||||
first.publish();
|
||||
const hit = prepareE2EShardReuse(request({ env }))!.lookup();
|
||||
expect(hit?.source.runId).toBe('1001/1');
|
||||
expect(prepareE2EShardReuse(request({ env: { ...env, EVALS_MODEL: 'changed' } }))!.lookup()).toBeNull();
|
||||
expect(prepareE2EShardReuse(request({ env: { ...env, EVALS_FRESH: '1' } }))).toBeNull();
|
||||
|
||||
const evalDir = path.join(scratch, 'evals');
|
||||
let launched = 0;
|
||||
const outcome = await runPaidShard([FILE], 1, 1, { rootDir: ROOT, logDir: scratch, evalDirBase: evalDir, env, log: () => {},
|
||||
expectedCaseIds: { [FILE]: ['setup-deploy-workflow'] },
|
||||
reuseFor: (files, childEnv) => prepareE2EShardReuse(request({ env: { ...childEnv } })),
|
||||
commandFor: () => { launched++; return { command: process.execPath, args: ['-e', 'process.exit(1)'] }; } });
|
||||
expect(launched).toBe(0);
|
||||
expect(outcome).toMatchObject({ status: 'passed', exitCode: 0, executedTests: 1, skippedTests: 0, reused: { runId: '1001/1' } });
|
||||
const recorded = JSON.parse(fs.readFileSync(path.join(evalDir, 'shards', 'skill-e2e-deploy', 'e2e-reused-skill-e2e-deploy.json'), 'utf8'));
|
||||
expect(recorded.tests).toEqual([expect.objectContaining({ name: 'setup-deploy-workflow', passed: true, execution: 'reused' })]);
|
||||
});
|
||||
|
||||
test('a failed shard never publishes a receipt', async () => {
|
||||
let published = 0;
|
||||
const outcome = await runPaidShard([FILE], 1, 1, { rootDir: ROOT, logDir: scratch, env: laneEnv(), log: () => {},
|
||||
reuseFor: () => ({ lookup: () => null, publish: () => { published++; } }),
|
||||
commandFor: () => ({ command: process.execPath, args: ['-e', 'process.exit(1)'] }) });
|
||||
expect(outcome.status).toBe('failed');
|
||||
expect(published).toBe(0);
|
||||
});
|
||||
|
||||
test('the report accepts reused results only in the fast PR profile', () => {
|
||||
const manifest = buildRunManifest({ tier: 'gate', sliceCount: 1, evalsAll: true, env: { EVALS_ALL: '1' } });
|
||||
const planned = manifest.entries.filter(entry => entry.status === 'planned');
|
||||
const reused = { inputKey: 'c'.repeat(64), runId: '1001/1', revision: 'd'.repeat(40), completedAt: 1 };
|
||||
const results: SliceResult[] = [{ version: 1, tier: 'gate', sliceIndex: 1, sliceCount: 1, outcomes: planned.map(entry => ({
|
||||
files: [entry.file], status: 'passed' as const, exitCode: 0, elapsedMs: 0, executedTests: 1, skippedTests: 0,
|
||||
...(entry.budget ? { budget: entry.budget } : {}), ...(entry.file === FILE ? { reused } : {}) })) }];
|
||||
expect(verifySliceResults(manifest, results).problems).toContain(`${FILE}: only the fast PR profile may reuse results; this lane executes fresh`);
|
||||
});
|
||||
});
|
||||
@@ -184,8 +184,8 @@ describe('E2E tier alignment (touchfiles declaration vs test self-gate)', () =>
|
||||
// Both self-gate shapes count: the raw predicate and the consolidated
|
||||
// helper (test/helpers/e2e-gate.ts documents this file as a consumer
|
||||
// that must recognize describeE2ETier/e2eTierEnabled).
|
||||
const selfGated = /EVALS_TIER\s*===\s*['"](gate|periodic)['"]/.test(content)
|
||||
|| /\b(?:describeE2ETier|e2eTierEnabled)\(\s*['"](gate|periodic)['"]/.test(content);
|
||||
const selfGated = /EVALS_TIER\s*===\s*['"](gate|periodic|marathon)['"]/.test(content)
|
||||
|| /\b(?:describeE2ETier|e2eTierEnabled)\(\s*['"](gate|periodic|marathon)['"]/.test(content);
|
||||
if (!usesNameSelection && selfGated) continue; // fail-open-safe standalone
|
||||
|
||||
invisible.push(
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
import { expect, test } from 'bun:test';
|
||||
import { resolvePaidShardBudget, retriesForFiles, planPaidShards, parseRunManifest, verifySliceResults, runPaidShard, buildRunManifest, paidShardWallUpperBoundMs, collectPaidTestFiles, selectPaidTestFiles, isOverlayTestFile, OVERLAY_MAX_ACTIVE_SHARDS, DEFAULT_SHARD_TIMEOUT_MS, DEFAULT_JOBS } from '../scripts/test-paid-shards';
|
||||
import { resolvePaidShardBudget, retriesForFiles, planPaidShards, parseRunManifest, verifySliceResults, runPaidShard, buildRunManifest, paidShardWallUpperBoundMs, collectPaidTestFiles, selectPaidTestFiles, isOverlayTestFile, DEFAULT_SHARD_TIMEOUT_MS, DEFAULT_JOBS, parseCliOptions, expandCaseShards, shardFile, sliceExecutionOrder, sliceSupervisedWallMs } from '../scripts/test-paid-shards';
|
||||
import { FINDING_RETRY_BUDGETS, ALL_TIERS, SHARD_RESERVE_MS } from './helpers/eval-budgets';
|
||||
import fs from 'node:fs';
|
||||
import os from 'node:os';
|
||||
@@ -8,7 +8,8 @@ import path from 'node:path';
|
||||
for (const budget of FINDING_RETRY_BUDGETS) {
|
||||
test(`${budget.file}: supervision preserves every existing attempt and retry`, () => {
|
||||
expect(budget.testMs).toBe(1_500_000);
|
||||
expect(budget.retries).toBe(1);
|
||||
// A 25-minute case is past RETRY_MAX_CASE_MS: a timed-out attempt is its verdict.
|
||||
expect(budget.retries).toBe(0);
|
||||
expect(retriesForFiles([budget.file])).toBe(budget.retries);
|
||||
expect(budget.shardReserveMs).toBe(SHARD_RESERVE_MS);
|
||||
expect(budget.shardMs).toBe(budget.cases * budget.testMs * (budget.retries + 1) + budget.shardReserveMs);
|
||||
@@ -24,9 +25,8 @@ for (const budget of FINDING_RETRY_BUDGETS) {
|
||||
expect([...source.matchAll(/timeoutMs:\s*1_500_000\b/g)]).toHaveLength(budget.cases);
|
||||
}
|
||||
expect([...source.matchAll(/1_500_000\s*\/\* physical ceiling:/g)]).toHaveLength(budget.cases);
|
||||
// Current periodic CI already supports this supervision wall.
|
||||
const workflow = Bun.YAML.parse(fs.readFileSync(path.join(import.meta.dir, '../.github/workflows/evals-periodic.yml'), 'utf8')) as any;
|
||||
expect(budget.shardMs).toBeLessThan(workflow.jobs['eval-slices']['timeout-minutes'] * 60_000);
|
||||
// The planned periodic CI job cap supports this supervision wall.
|
||||
expect(budget.shardMs).toBeLessThan(livePlan().plan!.ciTimeoutMinutes * 60_000);
|
||||
});
|
||||
|
||||
test(`${budget.file}: own-shard allocation leaves ordinary and explicit limits intact`, () => {
|
||||
@@ -104,31 +104,36 @@ test('actual shard launcher honors the explicit saved planner limit without a pr
|
||||
} finally { fs.rmSync(dir, { recursive: true, force: true }); }
|
||||
}, 10000);
|
||||
|
||||
const periodicWorkflow = Bun.YAML.parse(fs.readFileSync(path.join(import.meta.dir, '../.github/workflows/evals-periodic.yml'), 'utf8')) as any;
|
||||
const periodicJob = periodicWorkflow.jobs['eval-slices'];
|
||||
const periodicPlanStep = periodicWorkflow.jobs['plan-slices'].steps.find((step: any) => step.run?.includes('--tier periodic --emit-plan'));
|
||||
const periodicSliceCount = Number(periodicPlanStep.run.match(/--slices\s+(\d+)/)?.[1]);
|
||||
const periodicRunStep = periodicJob.steps.find((step: any) => step.run?.includes('--plan /tmp/paid-plan/manifest.json'));
|
||||
const periodicWorkers = Number(periodicRunStep.env.EVALS_JOBS);
|
||||
const livePlan = (discovered?: string[]) => buildRunManifest({ tier: 'periodic', sliceCount: periodicSliceCount,
|
||||
evalsAll: true, env: { EVALS_ALL: '1' }, discovered });
|
||||
function periodicLane() {
|
||||
const workflow = Bun.YAML.parse(fs.readFileSync(path.join(import.meta.dir, '../.github/workflows/evals-periodic.yml'), 'utf8')) as any;
|
||||
const job = workflow.jobs['eval-slices'];
|
||||
const planStep = workflow.jobs['plan-slices'].steps.find((step: any) => step.run?.includes('--tier periodic --emit-plan'));
|
||||
const planned = parseCliOptions(planStep.run.slice(planStep.run.indexOf('scripts/test-paid-shards.ts') + 'scripts/test-paid-shards.ts'.length).trim().split(/\s+/), {});
|
||||
const runStep = job.steps.find((step: any) => step.run?.includes('--plan /tmp/paid-plan/manifest.json'));
|
||||
return { job, planStep, planned, workers: Number(runStep.env.EVALS_JOBS) };
|
||||
}
|
||||
function livePlan(discovered?: string[]) {
|
||||
const { planned } = periodicLane();
|
||||
return buildRunManifest({ tier: 'periodic', sliceBudgetMs: planned.sliceBudgetMs!, jobs: planned.jobs,
|
||||
evalsAll: true, env: { EVALS_ALL: '1' }, discovered });
|
||||
}
|
||||
|
||||
test('live periodic census fits the declared CI wall including setup', () => {
|
||||
const { job, planStep, planned, workers } = periodicLane();
|
||||
const m = livePlan();
|
||||
expect(periodicPlanStep.run).not.toContain('--autoplan-slice');
|
||||
expect(periodicJob.strategy.matrix.slice).toEqual(Array.from({ length: periodicSliceCount }, (_, index) => index + 1));
|
||||
expect(periodicWorkers).toBe(2);
|
||||
const walls = Array.from({ length: periodicSliceCount }, (_, index) => {
|
||||
const files = m.entries.filter(e => e.status === 'planned' && e.slice === index + 1).map(e => e.file);
|
||||
const workers = files.some(isOverlayTestFile) ? Math.min(periodicWorkers, OVERLAY_MAX_ACTIVE_SHARDS) : periodicWorkers;
|
||||
return paidShardWallUpperBoundMs(files, workers);
|
||||
});
|
||||
expect(Math.max(...walls)).toBe(14_680_000);
|
||||
expect(periodicJob['timeout-minutes']).toBe(360);
|
||||
expect(periodicJob.strategy['max-parallel']).toBe(8);
|
||||
expect(Math.max(...walls) + 20 * 60_000).toBeLessThanOrEqual(periodicJob['timeout-minutes'] * 60_000);
|
||||
expect(m.entries.filter(e => e.status === 'planned')).toHaveLength(70);
|
||||
const overlays = m.entries.filter(e => e.status === 'planned' && e.slice === periodicSliceCount);
|
||||
expect(planStep.run).not.toContain('--autoplan-slice');
|
||||
expect(job.strategy.matrix.slice).toBe('${{ fromJSON(needs.plan-slices.outputs.periodic_slices) }}');
|
||||
expect(job['timeout-minutes']).toBe('${{ fromJSON(needs.plan-slices.outputs.periodic_timeout_minutes) }}');
|
||||
expect(workers).toBe(2);
|
||||
expect(planned.jobs).toBe(workers);
|
||||
const walls = Array.from({ length: m.sliceCount }, (_, index) => sliceSupervisedWallMs(sliceExecutionOrder(
|
||||
m.entries.filter(e => e.status === 'planned' && e.slice === index + 1)).map(e => e.file), workers));
|
||||
expect(Math.max(...walls) + 20 * 60_000).toBeLessThanOrEqual(m.plan!.ciTimeoutMinutes * 60_000);
|
||||
expect(m.plan!.ciTimeoutMinutes).toBeLessThanOrEqual(360);
|
||||
expect(m.sliceCount).toBeLessThanOrEqual(job.strategy['max-parallel']);
|
||||
const plannedFiles = new Set(m.entries.filter(e => e.status === 'planned').map(e => shardFile(e.file)));
|
||||
expect(plannedFiles).toEqual(new Set(selectPaidTestFiles(collectPaidTestFiles(), 'periodic').selected));
|
||||
const overlays = m.entries.filter(e => e.status === 'planned' && e.slice === m.sliceCount);
|
||||
expect(overlays).toHaveLength(4);
|
||||
expect(overlays.every(e => isOverlayTestFile(e.file))).toBe(true);
|
||||
});
|
||||
@@ -139,8 +144,8 @@ test('registered allocation is deterministic and preserves every discovered file
|
||||
expect(files).toContain('test/skill-e2e-ship-skip.test.ts');
|
||||
const m = livePlan(files);
|
||||
expect(livePlan([...files].reverse())).toEqual(m);
|
||||
expect(m.entries.map(e => e.file).sort()).toEqual([...files].sort());
|
||||
expect(new Set(m.entries.map(e => e.file)).size).toBe(files.length);
|
||||
expect([...new Set(m.entries.map(e => shardFile(e.file)))].sort()).toEqual([...files].sort());
|
||||
expect(new Set(m.entries.map(e => e.file)).size).toBe(m.entries.length);
|
||||
});
|
||||
|
||||
test('ordinary-only manifests retain round-robin allocation', () => {
|
||||
@@ -171,17 +176,18 @@ test('single-slice manifest retains all registered files with one allocation', (
|
||||
|
||||
test('current detach supervision covers the live-census floor', () => {
|
||||
const floorFor = (tier: 'gate' | 'periodic') => {
|
||||
const files = selectPaidTestFiles(collectPaidTestFiles(), tier).selected;
|
||||
// Case-sharded files contribute one shard per case, exactly as the runner plans.
|
||||
const files = expandCaseShards(selectPaidTestFiles(collectPaidTestFiles(), tier).selected, tier);
|
||||
const excess = files.reduce((n, file) => n + Math.max(0, resolvePaidShardBudget([file]).timeoutMs - DEFAULT_SHARD_TIMEOUT_MS), 0);
|
||||
return Math.ceil((Math.ceil(files.length / DEFAULT_JOBS) * DEFAULT_SHARD_TIMEOUT_MS + excess) / 1000 * 1.05);
|
||||
};
|
||||
const pkg = JSON.parse(fs.readFileSync(path.join(import.meta.dir, '../package.json'), 'utf8'));
|
||||
const periodicTimeout = Number(pkg.scripts['eval:bg:periodic'].match(/--timeout\s+(\d+)/)[1]);
|
||||
const gateTimeout = Number(pkg.scripts['eval:bg:gate'].match(/--timeout\s+(\d+)/)[1]);
|
||||
expect(floorFor('gate')).toBe(42_851);
|
||||
expect(floorFor('gate')).toBe(26_597);
|
||||
expect(gateTimeout).toBe(49_320);
|
||||
expect(gateTimeout).toBeGreaterThanOrEqual(floorFor('gate'));
|
||||
expect(floorFor('periodic')).toBe(37_727);
|
||||
expect(floorFor('periodic')).toBe(30_797);
|
||||
});
|
||||
|
||||
for (const jobs of [1, 2, 3]) test(`FIFO bound covers partial durations with ${jobs} workers`, () => {
|
||||
|
||||
@@ -24,9 +24,10 @@ import {
|
||||
DEFAULT_JOBS,
|
||||
DEFAULT_SHARD_TIMEOUT_MS,
|
||||
resolvePaidShardBudget,
|
||||
expandCaseShards,
|
||||
type PaidTier,
|
||||
} from '../scripts/test-paid-shards';
|
||||
import { FINDING_RETRY_BUDGETS } from './helpers/eval-budgets';
|
||||
import { FILE_RETRY_BUDGETS } from './helpers/eval-budgets';
|
||||
|
||||
const ROOT = path.resolve(import.meta.dir, '..');
|
||||
// 5% margin over the theoretical bound: detach setup, lock wait, aggregation.
|
||||
@@ -57,7 +58,7 @@ describe('eval:bg detach timeouts cover the sharded runner worst case', () => {
|
||||
['periodic', 'eval:bg:periodic'],
|
||||
] as Array<[PaidTier, string]>) {
|
||||
test(`${script} covers ordinary ${tier} waves plus registered excess x ${MARGIN}`, () => {
|
||||
const files = selectPaidTestFiles(collectPaidTestFiles(), tier).selected;
|
||||
const files = expandCaseShards(selectPaidTestFiles(collectPaidTestFiles(), tier).selected, tier);
|
||||
expect(files.length).toBeGreaterThan(0);
|
||||
const floor = Math.ceil(worstCaseSeconds(files) * MARGIN);
|
||||
const configured = detachTimeoutSeconds(script);
|
||||
@@ -78,7 +79,7 @@ describe('eval:bg detach timeouts cover the sharded runner worst case', () => {
|
||||
const pkg = JSON.parse(fs.readFileSync(path.join(ROOT, 'package.json'), 'utf8'));
|
||||
const jobs = Number(pkg.scripts['test:pr'].match(/EVALS_JOBS=\$\{EVALS_JOBS:-(\d+)\}/)?.[1]);
|
||||
expect(jobs).toBe(2);
|
||||
const files = selectPaidTestFiles(collectPaidTestFiles(), 'gate').selected;
|
||||
const files = expandCaseShards(selectPaidTestFiles(collectPaidTestFiles(), 'gate').selected, 'gate');
|
||||
const floor = Math.ceil(worstCaseSeconds(files, jobs) * MARGIN);
|
||||
expect(detachTimeoutSeconds('eval:bg:pr')).toBeGreaterThanOrEqual(floor);
|
||||
});
|
||||
@@ -86,7 +87,7 @@ describe('eval:bg detach timeouts cover the sharded runner worst case', () => {
|
||||
test('eval:bg:release covers both complete tiers and their existing margins', () => {
|
||||
const files = collectPaidTestFiles();
|
||||
const floor = (['gate', 'periodic'] as const).reduce((sum, tier) =>
|
||||
sum + Math.ceil(worstCaseSeconds(selectPaidTestFiles(files, tier).selected) * MARGIN), 0);
|
||||
sum + Math.ceil(worstCaseSeconds(expandCaseShards(selectPaidTestFiles(files, tier).selected, tier)) * MARGIN), 0);
|
||||
expect(detachTimeoutSeconds('eval:bg:release')).toBeGreaterThanOrEqual(floor);
|
||||
});
|
||||
});
|
||||
@@ -94,7 +95,9 @@ describe('eval:bg detach timeouts cover the sharded runner worst case', () => {
|
||||
// One long job and one ordinary job can run side by side; the long job still
|
||||
// needs its whole wall, regardless of the number of ordinary workers.
|
||||
test('a heterogeneous pair rejects the old uniform-wall floor', () => {
|
||||
const pair = [FINDING_RETRY_BUDGETS[0]!.file, 'test/skill-e2e-other.test.ts'];
|
||||
// The longest registered wall (single-attempt finding files now fit the ordinary wall).
|
||||
const longest = [...FILE_RETRY_BUDGETS].sort((a, b) => b.shardMs - a.shardMs)[0]!;
|
||||
const pair = [longest.file, 'test/skill-e2e-other.test.ts'];
|
||||
const actualLongest = Math.max(...pair.map(file => resolvePaidShardBudget([file]).timeoutMs)) / 1000;
|
||||
expect(worstCaseSeconds(pair, 2)).toBe(actualLongest);
|
||||
expect(worstCaseSeconds(pair, 2)).toBeGreaterThan(DEFAULT_SHARD_TIMEOUT_MS / 1000);
|
||||
|
||||
@@ -22,25 +22,53 @@
|
||||
import { describe, test, expect } from 'bun:test';
|
||||
import * as fs from 'fs';
|
||||
import * as path from 'path';
|
||||
import { buildRunManifest, parseCliOptions, isOverlayTestFile, OVERLAY_MAX_ACTIVE_SHARDS, paidShardWallUpperBoundMs } from '../scripts/test-paid-shards';
|
||||
import { buildRunManifest, parseCliOptions, sliceExecutionOrder, sliceSupervisedWallMs, CI_SETUP_ALLOWANCE_MINUTES } from '../scripts/test-paid-shards';
|
||||
|
||||
const ROOT = path.join(import.meta.dir, '..');
|
||||
const read = (rel: string) => fs.readFileSync(path.join(ROOT, rel), 'utf-8');
|
||||
|
||||
const evalsYml = read('.github/workflows/evals.yml');
|
||||
const periodicYml = read('.github/workflows/evals-periodic.yml');
|
||||
const marathonYml = read('.github/workflows/evals-marathon.yml');
|
||||
const registerAction = read('.github/actions/register-gstack-skills/action.yml');
|
||||
|
||||
/** Slice count the planner emits (`--slices N`) in a workflow source. */
|
||||
function plannedSlices(source: string): number[] {
|
||||
return [...source.matchAll(/--emit-plan\s+\S+\s+--slices\s+(\d+)/g)].map((m) => Number(m[1]));
|
||||
/** Every planner site: its manifest path and budget (`--slice-budget S --jobs J`). */
|
||||
function plannerSites(source: string): Array<{ manifest: string; budgetSeconds: number; jobs: number }> {
|
||||
return [...source.matchAll(/--emit-plan\s+(\S+)\s+--slice-budget\s+(\d+)\s+--jobs\s+(\d+)/g)]
|
||||
.map((m) => ({ manifest: m[1]!, budgetSeconds: Number(m[2]), jobs: Number(m[3]) }));
|
||||
}
|
||||
|
||||
/** The executor matrix's slice list (`slice: [1, 2, ...]`). */
|
||||
function matrixSlices(source: string): number[][] {
|
||||
return [...source.matchAll(/^\s+slice: \[([\d,\s]+)\]\s*$/gm)].map((m) =>
|
||||
m[1].split(',').map((n) => Number(n.trim())),
|
||||
);
|
||||
type Step = { id?: string; name?: string; run?: string; env?: Record<string, string>; with?: Record<string, string> };
|
||||
type Job = { needs?: string[]; env?: Record<string, string>; outputs?: Record<string, string>; 'timeout-minutes': string | number;
|
||||
strategy?: { 'max-parallel': number; matrix: { slice: string } }; steps: Step[] };
|
||||
|
||||
/**
|
||||
* An executor's matrix and timeout must come from the planner step that wrote
|
||||
* the manifest it downloads: `slices` from `[range(1; .sliceCount + 1)]` and
|
||||
* `timeout-minutes` from `.plan.ciTimeoutMinutes`, never hand-written numbers.
|
||||
*/
|
||||
function expectPlannedExecutor(source: string, executorName: string, prefix: string) {
|
||||
const workflow = Bun.YAML.parse(source) as { jobs: Record<string, Job> };
|
||||
const planner = workflow.jobs['plan-slices']!;
|
||||
const executor = workflow.jobs[executorName]!;
|
||||
expect(executor.needs).toContain('plan-slices');
|
||||
expect(executor.strategy!.matrix.slice).toBe(`\${{ fromJSON(needs.plan-slices.outputs.${prefix}slices) }}`);
|
||||
expect(executor['timeout-minutes']).toBe(`\${{ fromJSON(needs.plan-slices.outputs.${prefix}timeout_minutes) }}`);
|
||||
const [stepId] = /^\$\{\{ steps\.([\w-]+)\.outputs\.slices \}\}$/.exec(planner.outputs![`${prefix}slices`]!)!.slice(1);
|
||||
expect(planner.outputs![`${prefix}timeout_minutes`]).toBe(`\${{ steps.${stepId}.outputs.timeout_minutes }}`);
|
||||
const matrixStep = planner.steps.find(step => step.id === stepId)!;
|
||||
const manifest = /jq -c '\[range\(1; \.sliceCount \+ 1\)\]' (\S+)\)/.exec(matrixStep.run!)![1]!;
|
||||
expect(matrixStep.run).toContain(`jq -e '.plan.ciTimeoutMinutes' ${manifest})`);
|
||||
const emit = planner.steps.filter(step => step.run?.includes(`--emit-plan ${manifest} `));
|
||||
expect(emit).toHaveLength(1);
|
||||
const execute = executor.steps.filter(step => step.run?.includes('--plan '));
|
||||
expect(execute).toHaveLength(1);
|
||||
expect(execute[0]!.run).toContain(`--plan ${manifest} --slice \${{ matrix.slice }}`);
|
||||
expect(executor.steps.some(step => step.with?.path === manifest.replace(/\/manifest\.json$/, ''))).toBe(true);
|
||||
// The planner packs for exactly the executor's worker count.
|
||||
const site = plannerSites(emit[0]!.run!)[0]!;
|
||||
expect(execute[0]!.env?.EVALS_JOBS).toBe(String(site.jobs));
|
||||
return { site, emit: emit[0]!, execute: execute[0]!, executor, planner };
|
||||
}
|
||||
|
||||
describe('evals.yml sliced-lane wiring (post-matrix)', () => {
|
||||
@@ -67,13 +95,12 @@ describe('evals.yml sliced-lane wiring (post-matrix)', () => {
|
||||
expect(evalsYml).toMatch(/EVALS_TIER=gate bun --no-install run scripts\/test-paid-shards\.ts --tier gate --report /);
|
||||
});
|
||||
|
||||
test('executor matrix slice list matches the planner --slices count', () => {
|
||||
const planned = plannedSlices(evalsYml);
|
||||
const matrices = matrixSlices(evalsYml);
|
||||
expect(planned, 'expected exactly one --emit-plan site in evals.yml').toHaveLength(1);
|
||||
expect(matrices, 'expected exactly one slice matrix in evals.yml').toHaveLength(1);
|
||||
const n = planned[0];
|
||||
expect(matrices[0]).toEqual(Array.from({ length: n }, (_, i) => i + 1));
|
||||
test('executor matrix and timeout come from the one budget planner', () => {
|
||||
expect(plannerSites(evalsYml), 'expected exactly one --emit-plan site in evals.yml').toHaveLength(1);
|
||||
const { site } = expectPlannedExecutor(evalsYml, 'eval-slices', '');
|
||||
expect(site).toEqual({ manifest: '/tmp/paid-plan/manifest.json', budgetSeconds: 540, jobs: 2 });
|
||||
// The validation-phase planner writes the same manifest with the same budget.
|
||||
expect(evalsYml).toContain('sliceBudgetMs: 540000, jobs: 2');
|
||||
});
|
||||
|
||||
test('reconcile exit is captured via PIPESTATUS, never $? after a pipe', () => {
|
||||
@@ -81,7 +108,7 @@ describe('evals.yml sliced-lane wiring (post-matrix)', () => {
|
||||
// `$?` after `... | tee` is tee's exit — always 0. That made the
|
||||
// fail-closed reconcile gate silently fail-open (ship review army,
|
||||
// 2026-08-31). Both lanes must read PIPESTATUS[0].
|
||||
for (const [name, source] of [['evals.yml', evalsYml], ['evals-periodic.yml', periodicYml]] as const) {
|
||||
for (const [name, source] of [['evals.yml', evalsYml], ['evals-periodic.yml', periodicYml], ['evals-marathon.yml', marathonYml]] as const) {
|
||||
const reconcileBlocks = [...source.matchAll(/--report[^\n]*\| tee[^\n]*\n([\s\S]{0,400}?)GITHUB_OUTPUT/g)];
|
||||
expect(reconcileBlocks.length, `${name}: expected a tee'd reconcile step`).toBeGreaterThanOrEqual(1);
|
||||
for (const block of reconcileBlocks) {
|
||||
@@ -124,78 +151,89 @@ describe('evals.yml sliced-lane wiring (post-matrix)', () => {
|
||||
});
|
||||
|
||||
describe('evals-periodic.yml sliced-lane wiring', () => {
|
||||
test('the CI job cap covers the live periodic slice census plus setup', () => {
|
||||
type Env = Record<string, string>;
|
||||
const workflow = Bun.YAML.parse(periodicYml) as {
|
||||
env?: Env;
|
||||
jobs: Record<string, {
|
||||
env?: Env;
|
||||
'timeout-minutes': number;
|
||||
strategy?: { matrix: { slice: number[] } };
|
||||
steps: Array<{ run?: string; env?: Env }>;
|
||||
}>;
|
||||
};
|
||||
const planner = workflow.jobs['plan-slices'];
|
||||
const executor = workflow.jobs['eval-slices'];
|
||||
const plannerSteps = planner.steps.filter(step => step.run?.includes('EVALS_TIER=periodic ') && step.run.includes('--emit-plan '));
|
||||
const executorSteps = executor.steps.filter(step => step.run?.includes('--plan '));
|
||||
expect(plannerSteps).toHaveLength(1);
|
||||
expect(executorSteps).toHaveLength(1);
|
||||
const cliArgs = (run: string) => {
|
||||
const command = /\bbun(?: --no-install)? run scripts\/test-paid-shards\.ts /.exec(run);
|
||||
expect(command).not.toBeNull();
|
||||
return run.slice(command!.index + command![0].length)
|
||||
.replace(/\$\{\{\s*matrix\.slice\s*\}\}/g, '1').trim().split(/\s+/);
|
||||
};
|
||||
const plannerEnv = { ...workflow.env, ...planner.env, ...plannerSteps[0].env };
|
||||
const plannerOptions = parseCliOptions(cliArgs(plannerSteps[0].run!), plannerEnv);
|
||||
const executorOptions = parseCliOptions(cliArgs(executorSteps[0].run!), {
|
||||
...workflow.env, ...executor.env, ...executorSteps[0].env,
|
||||
});
|
||||
expect(plannerEnv.EVALS_ALL).toBe('1');
|
||||
expect(plannerOptions.tier).toBe('periodic');
|
||||
expect(executorOptions.tier).toBe('periodic');
|
||||
const slices = executor.strategy!.matrix.slice;
|
||||
expect(slices).toEqual(Array.from({ length: plannerOptions.slices }, (_, i) => i + 1));
|
||||
const manifest = buildRunManifest({
|
||||
tier: plannerOptions.tier, sliceCount: plannerOptions.slices,
|
||||
evalsAll: true, env: plannerEnv, rootDir: ROOT,
|
||||
});
|
||||
// Resolve the same per-file walls and overlay admission limit as execution.
|
||||
const explicitWall = executorOptions.timeoutExplicit ? executorOptions.timeoutMs : undefined;
|
||||
const setupAllowanceMinutes = 20;
|
||||
const allowances = slices.map(slice => {
|
||||
const files = manifest.entries.filter(entry => entry.status === 'planned' && entry.slice === slice).map(entry => entry.file);
|
||||
const normal = files.filter(file => !isOverlayTestFile(file));
|
||||
const overlay = files.filter(isOverlayTestFile);
|
||||
const bound = (group: string[], jobs: number) => paidShardWallUpperBoundMs(group, jobs, explicitWall);
|
||||
return (bound(normal, executorOptions.jobs) + bound(overlay, Math.min(executorOptions.jobs, OVERLAY_MAX_ACTIVE_SHARDS))) / 60_000;
|
||||
});
|
||||
expect(Math.max(...allowances)).toBeGreaterThan(0);
|
||||
const requiredMinutes = Math.max(...allowances) + setupAllowanceMinutes;
|
||||
expect(executor['timeout-minutes'],
|
||||
`periodic slice allowances ${allowances.join(', ')} minutes + ${setupAllowanceMinutes} minutes setup require ${requiredMinutes} CI minutes`,
|
||||
).toBeGreaterThanOrEqual(requiredMinutes);
|
||||
});
|
||||
const lanes = [
|
||||
{ source: periodicYml, name: 'evals-periodic.yml', executor: 'eval-slices', prefix: 'periodic_', tier: 'periodic' },
|
||||
{ source: periodicYml, name: 'evals-periodic.yml', executor: 'gate-census', prefix: 'gate_', tier: 'gate' },
|
||||
{ source: evalsYml, name: 'evals.yml', executor: 'eval-slices', prefix: '', tier: 'gate' },
|
||||
{ source: marathonYml, name: 'evals-marathon.yml', executor: 'eval-slices', prefix: '', tier: 'marathon' },
|
||||
] as const;
|
||||
|
||||
test('planner/executor/report tier=periodic and slice counts agree', () => {
|
||||
for (const lane of lanes) {
|
||||
test(`${lane.name}:${lane.executor} — the planned CI job cap covers every slice's supervised wall plus setup, and every slice starts at once`, () => {
|
||||
const { emit, execute, executor } = expectPlannedExecutor(lane.source, lane.executor, lane.prefix);
|
||||
const cliArgs = (run: string) => {
|
||||
const command = /\bbun(?: --no-install)? run scripts\/test-paid-shards\.ts /.exec(run);
|
||||
expect(command).not.toBeNull();
|
||||
return run.slice(command!.index + command![0].length)
|
||||
.replace(/\$\{\{\s*matrix\.slice\s*\}\}/g, '1').trim().split(/\s+/);
|
||||
};
|
||||
const workflow = Bun.YAML.parse(lane.source) as { env?: Record<string, string> };
|
||||
// The complete census (EVALS_ALL) is the largest plan any event can produce.
|
||||
const plannerEnv = { ...workflow.env, ...emit.env, EVALS_ALL: '1', EVALS_PROFILE: 'full' };
|
||||
const planned = parseCliOptions(cliArgs(emit.run!), plannerEnv);
|
||||
const active = parseCliOptions(cliArgs(execute.run!), { ...workflow.env, ...executor.env, ...execute.env, EVALS_PROFILE: 'full' });
|
||||
expect(planned.tier).toBe(lane.tier);
|
||||
expect(active.tier).toBe(lane.tier);
|
||||
expect(active.jobs).toBe(planned.jobs);
|
||||
const manifest = buildRunManifest({ tier: planned.tier, profile: 'full', sliceBudgetMs: planned.sliceBudgetMs!, jobs: planned.jobs,
|
||||
evalsAll: true, env: plannerEnv, rootDir: ROOT, skipJudges: planned.skipJudges });
|
||||
const walls = Array.from({ length: manifest.sliceCount }, (_, i) => sliceSupervisedWallMs(sliceExecutionOrder(
|
||||
manifest.entries.filter(entry => entry.status === 'planned' && entry.slice === i + 1)).map(entry => entry.file), planned.jobs));
|
||||
const requiredMinutes = Math.ceil(Math.max(0, ...walls) / 60_000) + CI_SETUP_ALLOWANCE_MINUTES;
|
||||
expect(CI_SETUP_ALLOWANCE_MINUTES).toBe(20);
|
||||
expect(manifest.plan!.ciTimeoutMinutes, `slice walls ${walls.join(', ')}ms`).toBe(requiredMinutes);
|
||||
// GitHub-hosted-style job ceiling: a plan past it must be split, not truncated.
|
||||
expect(manifest.plan!.ciTimeoutMinutes).toBeLessThanOrEqual(360);
|
||||
expect(manifest.sliceCount, `${lane.name}:${lane.executor} plans more slices than max-parallel starts at once`)
|
||||
.toBeLessThanOrEqual(executor.strategy!['max-parallel']);
|
||||
});
|
||||
}
|
||||
|
||||
test('planner/executor/report tier=periodic agree and plan with the ~9-minute budget', () => {
|
||||
expect(periodicYml).toMatch(/EVALS_TIER=periodic bun --no-install run scripts\/test-paid-shards\.ts --tier periodic --emit-plan/);
|
||||
expect(periodicYml).toMatch(/EVALS_TIER=periodic bun run scripts\/test-paid-shards\.ts --tier periodic --plan .* --slice /);
|
||||
expect(periodicYml).toMatch(/EVALS_TIER=periodic bun --no-install run scripts\/test-paid-shards\.ts --tier periodic --report /);
|
||||
const planned = plannedSlices(periodicYml);
|
||||
const matrices = matrixSlices(periodicYml);
|
||||
// Periodic work and the full gate census have distinct immutable plans.
|
||||
expect(planned).toHaveLength(2);
|
||||
expect(matrices).toHaveLength(2);
|
||||
for (const [index, count] of planned.entries()) {
|
||||
expect(matrices[index]).toEqual(Array.from({ length: count }, (_, i) => i + 1));
|
||||
expect(plannerSites(periodicYml)).toEqual([
|
||||
{ manifest: '/tmp/paid-plan/manifest.json', budgetSeconds: 540, jobs: 2 },
|
||||
{ manifest: '/tmp/gate-census-plan/manifest.json', budgetSeconds: 540, jobs: 2 },
|
||||
]);
|
||||
});
|
||||
});
|
||||
|
||||
describe('evals-marathon.yml non-blocking lane', () => {
|
||||
const workflow = Bun.YAML.parse(marathonYml) as { on: Record<string, unknown>; env: Record<string, string>; jobs: Record<string, Job> };
|
||||
|
||||
test('runs weekly and on dispatch, always fresh, with its own fail-closed report and tracking issue', () => {
|
||||
expect(Object.keys(workflow.on).sort()).toEqual(['schedule', 'workflow_dispatch']);
|
||||
expect(workflow.env).toMatchObject({ EVALS_PROFILE: 'full', EVALS_FRESH: '1', EVALS_CACHE_PURPOSE: 'marathon' });
|
||||
expect(marathonYml).not.toContain('actions/cache');
|
||||
expect(marathonYml).toMatch(/EVALS_TIER=marathon bun --no-install run scripts\/test-paid-shards\.ts --tier marathon --emit-plan \/tmp\/marathon-plan\/manifest\.json --slice-budget 1 --jobs 1/);
|
||||
expect(marathonYml).toMatch(/EVALS_TIER=marathon bun run scripts\/test-paid-shards\.ts --tier marathon --plan .* --slice /);
|
||||
const report = workflow.jobs.report!;
|
||||
expect(report.needs).toEqual(['plan-slices', 'eval-slices']);
|
||||
const reconcile = report.steps.find(step => step.id === 'reconcile')!;
|
||||
expect(reconcile.run).toContain('EVALS_TIER=marathon bun --no-install run scripts/test-paid-shards.ts --tier marathon --report /tmp/marathon-report');
|
||||
const guards = report.steps.filter(step => /Upsert tracking|Fail the workflow/.test(step.name ?? ''));
|
||||
expect(guards).toHaveLength(2);
|
||||
for (const step of guards) {
|
||||
expect((step as { if?: string }).if).toContain("steps.reconcile.outputs.exit != '0'");
|
||||
expect((step as { if?: string }).if).toContain("needs.eval-slices.result != 'success'");
|
||||
}
|
||||
expect(marathonYml).toContain('Weekly marathon evals: red lane needs triage');
|
||||
});
|
||||
|
||||
test('the blocking lanes never plan or execute the marathon tier', () => {
|
||||
for (const source of [evalsYml, periodicYml]) {
|
||||
expect(source).not.toContain('--tier marathon');
|
||||
expect(source).not.toContain('EVALS_TIER=marathon');
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
describe('shared setup composites (both surviving lanes)', () => {
|
||||
test('both lanes register skills through the shared composite', () => {
|
||||
for (const [name, source] of [['evals.yml', evalsYml], ['evals-periodic.yml', periodicYml]] as const) {
|
||||
describe('shared setup composites (every paid lane)', () => {
|
||||
test('every lane registers skills through the shared composite', () => {
|
||||
for (const [name, source] of [['evals.yml', evalsYml], ['evals-periodic.yml', periodicYml], ['evals-marathon.yml', marathonYml]] as const) {
|
||||
expect(source, `${name} must use the register-gstack-skills composite`)
|
||||
.toContain('uses: ./.github/actions/register-gstack-skills');
|
||||
// No inline re-implementation creeping back beside the composite.
|
||||
@@ -218,7 +256,7 @@ describe('shared setup composites (both surviving lanes)', () => {
|
||||
for (const action of ['seed-claude-config', 'restore-deps', 'fix-bun-temp']) {
|
||||
expect(fs.existsSync(path.join(ROOT, '.github', 'actions', action, 'action.yml')), `missing composite: ${action}`).toBe(true);
|
||||
}
|
||||
for (const [name, source] of [['evals.yml', evalsYml], ['evals-periodic.yml', periodicYml]] as const) {
|
||||
for (const [name, source] of [['evals.yml', evalsYml], ['evals-periodic.yml', periodicYml], ['evals-marathon.yml', marathonYml]] as const) {
|
||||
expect(source, `${name} must use seed-claude-config`).toContain('uses: ./.github/actions/seed-claude-config');
|
||||
expect(source, `${name} must use restore-deps`).toContain('uses: ./.github/actions/restore-deps');
|
||||
expect(source, `${name} must use fix-bun-temp`).toContain('uses: ./.github/actions/fix-bun-temp');
|
||||
|
||||
+103
-40
@@ -46,70 +46,133 @@ export const ALL_TIERS = {
|
||||
/** Supervision reserve added to every registered whole-file wall. */
|
||||
export const SHARD_RESERVE_MS = 2 * 60_000;
|
||||
|
||||
/** Whole-file supervision must cover each existing attempt and its retry.
|
||||
* These fixtures already allow 25 minutes per case; the old 30-minute
|
||||
* wall could kill a second attempt after five minutes. No case budget grows.
|
||||
/**
|
||||
* Retry policy (approved 2026-09-29): a timed-out attempt is a verdict. Bun's
|
||||
* --retry reruns a failed case after it may have spent its whole budget, so an
|
||||
* automatic retry is kept only where one more attempt is short: every case of
|
||||
* the file has a per-attempt budget of at most RETRY_MAX_CASE_MS, the CAPTURE
|
||||
* tier plus its recording grace. Those failures are fast flake classes (API
|
||||
* blips, tool hiccups) and a retry costs at most one more short attempt. Files
|
||||
* with any longer case run once. Per-case budgets never change with this rule.
|
||||
*/
|
||||
export const RETRY_MAX_CASE_MS = CAPTURE_MS + 15_000;
|
||||
|
||||
export function retriesWithinCaseCap(caseMs: number, configuredRetries: number): number {
|
||||
return caseMs <= RETRY_MAX_CASE_MS ? configuredRetries : 0;
|
||||
}
|
||||
|
||||
/**
|
||||
* Unregistered paid files that keep one automatic retry: every case budget is
|
||||
* JUDGE or CAPTURE tier (test/paid-retry-supervision.test.ts scans each source).
|
||||
* Registered rows below derive retries from their declared caseMs; every other
|
||||
* paid file runs once.
|
||||
*/
|
||||
export const SHORT_CASE_RETRY_FILES: readonly string[] = [
|
||||
'test/codex-e2e-sol-scope.test.ts',
|
||||
'test/llm-judge-recommendation.test.ts',
|
||||
'test/skill-e2e-ask-user-question-format-compliance.test.ts',
|
||||
'test/skill-e2e-benchmark-providers.test.ts',
|
||||
'test/skill-e2e-bws.test.ts',
|
||||
'test/skill-e2e-context-skills.test.ts',
|
||||
'test/skill-e2e-coverage-audit.test.ts',
|
||||
'test/skill-e2e-diagram.test.ts',
|
||||
'test/skill-e2e-first-task-scaffold.test.ts',
|
||||
'test/skill-e2e-gbrain-roundtrip-local.test.ts',
|
||||
'test/skill-e2e-hermetic-canary.test.ts',
|
||||
'test/skill-e2e-investigate-owned-completion.test.ts',
|
||||
'test/skill-e2e-investigate-owned-termination.test.ts',
|
||||
'test/skill-e2e-learnings.test.ts',
|
||||
'test/skill-e2e-plan-tune.test.ts',
|
||||
'test/skill-e2e-qa-functional-fix.test.ts',
|
||||
'test/skill-e2e-qa-functional.test.ts',
|
||||
'test/skill-e2e-review-army.test.ts',
|
||||
'test/skill-e2e-review.test.ts',
|
||||
'test/skill-e2e-session-intelligence.test.ts',
|
||||
'test/skill-e2e-setup-gbrain-bad-token.test.ts',
|
||||
'test/skill-e2e-setup-gbrain-path4-local-pglite.test.ts',
|
||||
'test/skill-e2e-setup-gbrain-remote.test.ts',
|
||||
'test/skill-e2e-ship-hook-consent.test.ts',
|
||||
'test/skill-e2e-ship-hook-refresh.test.ts',
|
||||
'test/skill-e2e-ship-skip.test.ts',
|
||||
'test/skill-e2e-sync-gbrain-readiness.test.ts',
|
||||
'test/skill-e2e-third-party-actions.test.ts',
|
||||
'test/skill-e2e-triage.test.ts',
|
||||
'test/skill-routing-e2e.test.ts',
|
||||
];
|
||||
|
||||
/** Whole-file supervision covers every attempt the retry policy allows.
|
||||
* These fixtures allow 25 minutes per case, so they run once.
|
||||
* Reserve the sequential upper bound even when Bun runs sibling cases together.
|
||||
*/
|
||||
export const FINDING_RETRY_BUDGETS = [
|
||||
{ file: 'test/skill-e2e-plan-ceo-split-overflow.test.ts', cases: 1 },
|
||||
{ file: 'test/skill-e2e-plan-eng-multi-finding-batching.test.ts', cases: 1 },
|
||||
].map(({ file, cases }) => ({
|
||||
file, cases,
|
||||
id: `${file.slice('test/skill-e2e-'.length, -'.test.ts'.length)}-existing-retry-v1`,
|
||||
testMs: 1_500_000,
|
||||
retries: 1,
|
||||
shardReserveMs: SHARD_RESERVE_MS,
|
||||
shardMs: cases * 1_500_000 * 2 + SHARD_RESERVE_MS,
|
||||
}));
|
||||
].map(({ file, cases }) => {
|
||||
const retries = retriesWithinCaseCap(1_500_000, 1);
|
||||
return {
|
||||
file, cases,
|
||||
id: `${file.slice('test/skill-e2e-'.length, -'.test.ts'.length)}-existing-retry-v1`,
|
||||
testMs: 1_500_000,
|
||||
caseMs: 1_500_000,
|
||||
retries,
|
||||
shardReserveMs: SHARD_RESERVE_MS,
|
||||
shardMs: cases * 1_500_000 * (retries + 1) + SHARD_RESERVE_MS,
|
||||
};
|
||||
});
|
||||
|
||||
/** Three existing captures and one configured retry; only supervision grows. */
|
||||
/** Three existing captures in one 16-minute case, so the file runs once. */
|
||||
export const AUQ_CONSISTENCY_RETRY_BUDGET = {
|
||||
file: 'test/skill-e2e-auq-consistency.test.ts',
|
||||
id: 'auq-consistency-existing-retry-v1',
|
||||
cases: 1,
|
||||
testMs: 3 * CAPTURE_MS + 60_000,
|
||||
retries: 1,
|
||||
caseMs: 3 * CAPTURE_MS + 60_000,
|
||||
retries: retriesWithinCaseCap(3 * CAPTURE_MS + 60_000, 1),
|
||||
shardReserveMs: SHARD_RESERVE_MS,
|
||||
shardMs: (3 * CAPTURE_MS + 60_000) * 2 + SHARD_RESERVE_MS,
|
||||
shardMs: (3 * CAPTURE_MS + 60_000) * (retriesWithinCaseCap(3 * CAPTURE_MS + 60_000, 1) + 1) + SHARD_RESERVE_MS,
|
||||
} as const;
|
||||
|
||||
/** These fixtures have a fixed case count in every supported tier. */
|
||||
export const STRICT_RETRY_CASE_BUDGETS = [...FINDING_RETRY_BUDGETS, AUQ_CONSISTENCY_RETRY_BUDGET];
|
||||
|
||||
/** Whole-file walls cover all existing cases and retries, even if Bun runs them
|
||||
* sequentially. Mixed-tier files reserve their larger complete tier, never a
|
||||
* currently selected subset. These rows add no case-count or model-work policy.
|
||||
* The 10-second terms preserve the existing Codex/recording finalization grace.
|
||||
/** Whole-file walls cover all existing cases and every allowed attempt, even if
|
||||
* Bun runs them sequentially. Mixed-tier files reserve their larger complete
|
||||
* tier, never a currently selected subset. caseMs is the longest single case
|
||||
* budget, which decides the retry (RETRY_MAX_CASE_MS). These rows add no
|
||||
* case-count or model-work policy. The 10-second terms preserve the existing
|
||||
* Codex/recording finalization grace.
|
||||
*/
|
||||
export const FILE_RETRY_BUDGETS = [
|
||||
...STRICT_RETRY_CASE_BUDGETS,
|
||||
...[
|
||||
{ file: 'test/skill-e2e-qa-callers.test.ts', attemptMs: 5 * (CAPTURE_MS + 15_000), retries: 1 },
|
||||
{ file: 'test/skill-e2e-shared-libs-paths.test.ts', attemptMs: 3 * CAPTURE_LONG_MS, retries: 1 },
|
||||
{ file: 'test/skill-e2e-ship-docsync.test.ts', attemptMs: 5 * CAPTURE_LONG_MS + 8 * CAPTURE_MS, retries: 1 },
|
||||
{ file: 'test/skill-e2e-qa-callers.test.ts', attemptMs: 5 * (CAPTURE_MS + 15_000), caseMs: CAPTURE_MS + 15_000, configuredRetries: 1 },
|
||||
{ file: 'test/skill-e2e-shared-libs-paths.test.ts', attemptMs: 3 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 1 },
|
||||
{ file: 'test/skill-e2e-ship-docsync.test.ts', attemptMs: 5 * CAPTURE_LONG_MS + 8 * CAPTURE_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 1 },
|
||||
// Seventeen workflow judges include their 10s recording grace; the other
|
||||
// seven judges retain 120s. Supervise all 24 and the existing one retry.
|
||||
{ file: 'test/skill-llm-eval.test.ts', attemptMs: 17 * (JUDGE_MS + 10_000) + 7 * JUDGE_MS, retries: 1 },
|
||||
{ file: 'test/skill-e2e-auq-matrix.test.ts', attemptMs: 6 * CAPTURE_MS, retries: 1 },
|
||||
{ file: 'test/skill-e2e-plan-format.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), retries: 1 },
|
||||
{ file: 'test/skill-e2e-auto-decide-preserved.test.ts', attemptMs: PTY_MS, retries: 1 },
|
||||
{ file: 'test/skill-e2e-plan-ceo-finding-floor.test.ts', attemptMs: PTY_MS, retries: 1 },
|
||||
{ file: 'test/skill-e2e-plan-eng-finding-floor.test.ts', attemptMs: PTY_MS, retries: 1 },
|
||||
{ file: 'test/skill-e2e-plan-design-finding-floor.test.ts', attemptMs: PTY_MS, retries: 1 },
|
||||
{ file: 'test/skill-e2e-plan-devex-finding-floor.test.ts', attemptMs: PTY_MS, retries: 1 },
|
||||
{ file: 'test/skill-e2e-plan-mode-no-op.test.ts', attemptMs: 5 * CAPTURE_LONG_MS, retries: 2 },
|
||||
{ file: 'test/skill-e2e-plan-ceo-mode-routing.test.ts', attemptMs: 2 * CAPTURE_LONG_MS, retries: 1 },
|
||||
{ file: 'test/skill-e2e-plan-eng-plan-mode.test.ts', attemptMs: 2 * CAPTURE_LONG_MS, retries: 1 },
|
||||
{ file: 'test/skill-e2e-plan-prosons.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), retries: 1 },
|
||||
{ file: 'test/skill-llm-eval.test.ts', attemptMs: 17 * (JUDGE_MS + 10_000) + 7 * JUDGE_MS, caseMs: JUDGE_MS + 10_000, configuredRetries: 1 },
|
||||
{ file: 'test/skill-e2e-auq-matrix.test.ts', attemptMs: 6 * CAPTURE_MS, caseMs: CAPTURE_MS, configuredRetries: 1 },
|
||||
{ file: 'test/skill-e2e-plan-format.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), caseMs: CAPTURE_MS + 10_000, configuredRetries: 1 },
|
||||
{ file: 'test/skill-e2e-auto-decide-preserved.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 },
|
||||
{ file: 'test/skill-e2e-plan-ceo-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 },
|
||||
{ file: 'test/skill-e2e-plan-eng-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 },
|
||||
{ file: 'test/skill-e2e-plan-design-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 },
|
||||
{ file: 'test/skill-e2e-plan-devex-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 },
|
||||
{ file: 'test/skill-e2e-plan-mode-no-op.test.ts', attemptMs: 5 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 2 },
|
||||
{ file: 'test/skill-e2e-plan-ceo-mode-routing.test.ts', attemptMs: 2 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 1 },
|
||||
{ file: 'test/skill-e2e-plan-eng-plan-mode.test.ts', attemptMs: 2 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 1 },
|
||||
{ file: 'test/skill-e2e-plan-prosons.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), caseMs: CAPTURE_MS + 10_000, configuredRetries: 1 },
|
||||
// Gate: six 300s cases + one 610s case; periodic: two 900s + three 600s.
|
||||
{ file: 'test/skill-e2e-plan.test.ts', attemptMs: Math.max(6 * CAPTURE_MS + CAPTURE_LONG_MS + 10_000, 2 * PTY_MS + 3 * CAPTURE_LONG_MS), retries: 1 },
|
||||
].map(({ file, attemptMs, retries }) => ({
|
||||
file, attemptMs, retries,
|
||||
id: `${file.slice('test/'.length, -'.test.ts'.length)}-existing-retry-v1`,
|
||||
shardReserveMs: SHARD_RESERVE_MS,
|
||||
shardMs: attemptMs * (retries + 1) + SHARD_RESERVE_MS,
|
||||
})),
|
||||
{ file: 'test/skill-e2e-plan.test.ts', attemptMs: Math.max(6 * CAPTURE_MS + CAPTURE_LONG_MS + 10_000, 2 * PTY_MS + 3 * CAPTURE_LONG_MS), caseMs: PTY_MS, configuredRetries: 1 },
|
||||
].map(({ file, attemptMs, caseMs, configuredRetries }) => {
|
||||
const retries = retriesWithinCaseCap(caseMs, configuredRetries);
|
||||
return {
|
||||
file, attemptMs, caseMs, retries,
|
||||
id: `${file.slice('test/'.length, -'.test.ts'.length)}-existing-retry-v1`,
|
||||
shardReserveMs: SHARD_RESERVE_MS,
|
||||
shardMs: attemptMs * (retries + 1) + SHARD_RESERVE_MS,
|
||||
};
|
||||
}),
|
||||
];
|
||||
|
||||
/** No paid test may exceed the ordinary tiers; arbitrary per-file escapes fail. */
|
||||
|
||||
@@ -22,17 +22,18 @@ const fakeEnv = {
|
||||
|
||||
describe('overlay file policy', () => {
|
||||
test('grouped planning isolates every overlay and preserves ordinary retries', () => {
|
||||
const workflow = 'test/skill-e2e-workflow.test.ts';
|
||||
const files = [...overlayFiles, normalFile, workflow];
|
||||
// Two short-case files keep their one retry (timeout-is-a-verdict rule).
|
||||
const workflow = 'test/skill-e2e-review.test.ts';
|
||||
const files = [...overlayFiles, 'test/skill-e2e-triage.test.ts', workflow];
|
||||
for (const maxFilesPerShard of [2, 3, 10]) {
|
||||
const shards = planPaidShards(files, { maxFilesPerShard });
|
||||
expect(shards.flat().sort()).toEqual([...files].sort());
|
||||
for (const file of overlayFiles) expect(shards).toContainEqual([file]);
|
||||
const workflowShard = shards.find(shard => shard.includes(workflow))!;
|
||||
expect(workflowShard.some(isOverlayTestFile)).toBe(false);
|
||||
expect(retriesForFiles(workflowShard)).toBe(2);
|
||||
expect(retriesForFiles(workflowShard)).toBe(1);
|
||||
const args = buildPaidShardArgs(workflowShard, resolvePaidShardTimeoutMs(workflowShard), 2, retriesForFiles(workflowShard));
|
||||
expect(args[args.indexOf('--retry') + 1]).toBe('2');
|
||||
expect(args[args.indexOf('--retry') + 1]).toBe('1');
|
||||
expect(planPaidShards(files.map(file => file.replaceAll('/', '\\')), { maxFilesPerShard })).toEqual(shards);
|
||||
}
|
||||
});
|
||||
@@ -65,8 +66,10 @@ describe('overlay file policy', () => {
|
||||
for (const file of [normalFile, 'test/skill-e2e-overlay-harness.test.ts', 'test/model-overlays.test.ts']) {
|
||||
expect(isOverlayTestFile(file)).toBe(false);
|
||||
expect(resolvePaidShardTimeoutMs([file])).toBe(DEFAULT_SHARD_TIMEOUT_MS);
|
||||
expect(retriesForFiles([file])).toBe(1);
|
||||
// Not overlays; unlisted files run once because their case budget is unknown.
|
||||
expect(retriesForFiles([file])).toBe(0);
|
||||
}
|
||||
expect(retriesForFiles(['test/skill-e2e-review.test.ts'])).toBe(1);
|
||||
expect(resolvePaidShardTimeoutMs([normalFile], 1234)).toBe(1234);
|
||||
expect(resolvePaidShardTimeoutMs([overlayFiles[0]], 1_900_000)).toBe(1_900_000);
|
||||
expect(() => resolvePaidShardTimeoutMs([overlayFiles[0]], 1_800_000)).toThrow('explicit wall');
|
||||
@@ -157,8 +160,8 @@ describe('overlay manifest affinity and CI capacity', () => {
|
||||
const workflow = Bun.YAML.parse(fs.readFileSync(path.join(ROOT, '.github/workflows/evals-periodic.yml'), 'utf8')) as {
|
||||
jobs: Record<string, {
|
||||
steps: Array<{ run?: string; env?: NodeJS.ProcessEnv }>;
|
||||
strategy: { matrix: { slice: number[] } };
|
||||
'timeout-minutes': number;
|
||||
strategy: { matrix: { slice: string } };
|
||||
'timeout-minutes': string;
|
||||
}>;
|
||||
};
|
||||
const job = workflow.jobs['eval-slices'];
|
||||
@@ -166,13 +169,20 @@ describe('overlay manifest affinity and CI capacity', () => {
|
||||
const jobs = parseCliOptions([], step.env).jobs;
|
||||
expect(jobs).toBe(2);
|
||||
expect(parseCliOptions([], step.env).withinShardConcurrency).toBe(2);
|
||||
expect(job.strategy.matrix.slice).toEqual([1, 2, 3, 4, 5, 6, 7]);
|
||||
expect(job.strategy.matrix.slice).toBe('${{ fromJSON(needs.plan-slices.outputs.periodic_slices) }}');
|
||||
expect(job['timeout-minutes']).toBe('${{ fromJSON(needs.plan-slices.outputs.periodic_timeout_minutes) }}');
|
||||
const normalMinutes = Math.ceil(18 / jobs) * resolvePaidShardTimeoutMs([normalFiles[0]]) / 60_000;
|
||||
const overlayMinutes = Math.ceil(overlayFiles.length / OVERLAY_MAX_ACTIVE_SHARDS)
|
||||
* Math.max(...overlayFiles.map(file => resolvePaidShardTimeoutMs([file]))) / 60_000;
|
||||
expect(normalMinutes).toBe(270);
|
||||
expect(overlayMinutes).toBe(122);
|
||||
expect(job['timeout-minutes']).toBeGreaterThanOrEqual(Math.max(normalMinutes, overlayMinutes) + 20);
|
||||
// The CI budget plan (what the workflow runs) keeps the four overlays
|
||||
// in one final one-at-a-time slice and its job cap covers them.
|
||||
const budget = buildRunManifest({ ...opts, sliceCount: undefined, sliceBudgetMs: 540_000, jobs });
|
||||
const lastSlice = budget.entries.filter(e => e.slice === budget.sliceCount).map(e => e.file).sort();
|
||||
expect(lastSlice).toEqual([...overlayFiles].sort());
|
||||
expect(budget.plan!.ciTimeoutMinutes).toBeGreaterThanOrEqual(overlayMinutes + 20);
|
||||
expect(budget.plan!.ciTimeoutMinutes).toBe(Math.ceil(overlayMinutes) + 20);
|
||||
|
||||
// Gate selection keeps its original periodic exclusion and all six
|
||||
// ordinary slices; reservation does not spend an empty slot in gate.
|
||||
|
||||
@@ -150,7 +150,8 @@ describe('PR profile paid-runner integration', () => {
|
||||
expect(() => parseRunManifest(JSON.stringify(injected))).toThrow('outside its PR case selection');
|
||||
for (const action of ['remove', 'skip', 'duplicate'] as const) {
|
||||
const missing = structuredClone(manifest);
|
||||
const file = 'test/skill-e2e-plan.test.ts';
|
||||
// plan.test is case-sharded: its PR case runs as `<file>#<case id>`.
|
||||
const file = manifest.entries.find(entry => entry.status === 'planned' && entry.file.startsWith('test/skill-e2e-plan.test.ts#'))!.file;
|
||||
if (action === 'remove') missing.entries = missing.entries.filter(entry => entry.file !== file);
|
||||
if (action === 'skip') missing.entries.find(entry => entry.file === file)!.status = 'skipped-by-diff';
|
||||
if (action === 'duplicate') missing.entries.push({ ...missing.entries.find(entry => entry.file === file)! });
|
||||
|
||||
@@ -4,34 +4,69 @@ import { join } from 'node:path';
|
||||
import {
|
||||
buildPaidShardArgs, buildRunManifest, parseRunManifest, planPaidShards,
|
||||
DEFAULT_JOBS, parseCliOptions, paidShardWallUpperBoundMs, resolvePaidShardBudget, retriesForFiles, verifySliceResults, collectPaidTestFiles, selectPaidTestFiles,
|
||||
shardFile, sliceExecutionOrder, sliceSupervisedWallMs, CASE_SHARDED_FILES,
|
||||
} from '../scripts/test-paid-shards';
|
||||
import {
|
||||
ALL_TIERS, AUQ_CONSISTENCY_RETRY_BUDGET, FILE_RETRY_BUDGETS,
|
||||
ALL_TIERS, AUQ_CONSISTENCY_RETRY_BUDGET, FILE_RETRY_BUDGETS, RETRY_MAX_CASE_MS, SHORT_CASE_RETRY_FILES,
|
||||
FINDING_RETRY_BUDGETS, STRICT_RETRY_CASE_BUDGETS,
|
||||
} from './helpers/eval-budgets';
|
||||
|
||||
import { E2E_TOUCHFILES } from './helpers/touchfiles';
|
||||
|
||||
const read = (file: string) => readFileSync(join(import.meta.dir, '..', file), 'utf8');
|
||||
const newBudgets = FILE_RETRY_BUDGETS.filter(row => !FINDING_RETRY_BUDGETS.some(old => old.file === row.file));
|
||||
// Walls cover every attempt the retry rule allows: files with a case budget
|
||||
// past RETRY_MAX_CASE_MS run once (a timed-out attempt is a verdict).
|
||||
const expectedWalls = {
|
||||
'test/skill-e2e-qa-callers.test.ts': 3_270_000,
|
||||
'test/skill-e2e-shared-libs-paths.test.ts': 3_720_000,
|
||||
'test/skill-e2e-ship-docsync.test.ts': 10_920_000,
|
||||
'test/skill-e2e-shared-libs-paths.test.ts': 1_920_000,
|
||||
'test/skill-e2e-ship-docsync.test.ts': 5_520_000,
|
||||
'test/skill-llm-eval.test.ts': 6_220_000,
|
||||
'test/skill-e2e-auq-consistency.test.ts': 2_040_000,
|
||||
'test/skill-e2e-auq-consistency.test.ts': 1_080_000,
|
||||
'test/skill-e2e-auq-matrix.test.ts': 3_720_000,
|
||||
'test/skill-e2e-plan-format.test.ts': 2_600_000,
|
||||
'test/skill-e2e-auto-decide-preserved.test.ts': 1_920_000,
|
||||
'test/skill-e2e-plan-ceo-finding-floor.test.ts': 1_920_000,
|
||||
'test/skill-e2e-plan-eng-finding-floor.test.ts': 1_920_000,
|
||||
'test/skill-e2e-plan-design-finding-floor.test.ts': 1_920_000,
|
||||
'test/skill-e2e-plan-devex-finding-floor.test.ts': 1_920_000,
|
||||
'test/skill-e2e-plan-mode-no-op.test.ts': 9_120_000,
|
||||
'test/skill-e2e-plan-ceo-mode-routing.test.ts': 2_520_000,
|
||||
'test/skill-e2e-plan-eng-plan-mode.test.ts': 2_520_000,
|
||||
'test/skill-e2e-auto-decide-preserved.test.ts': 1_020_000,
|
||||
'test/skill-e2e-plan-ceo-finding-floor.test.ts': 1_020_000,
|
||||
'test/skill-e2e-plan-eng-finding-floor.test.ts': 1_020_000,
|
||||
'test/skill-e2e-plan-design-finding-floor.test.ts': 1_020_000,
|
||||
'test/skill-e2e-plan-devex-finding-floor.test.ts': 1_020_000,
|
||||
'test/skill-e2e-plan-mode-no-op.test.ts': 3_120_000,
|
||||
'test/skill-e2e-plan-ceo-mode-routing.test.ts': 1_320_000,
|
||||
'test/skill-e2e-plan-eng-plan-mode.test.ts': 1_320_000,
|
||||
'test/skill-e2e-plan-prosons.test.ts': 2_600_000,
|
||||
'test/skill-e2e-plan.test.ts': 7_320_000,
|
||||
'test/skill-e2e-plan.test.ts': 3_720_000,
|
||||
};
|
||||
|
||||
test('retry rule: only files whose every case is CAPTURE tier or shorter retry; longer cases run once', () => {
|
||||
expect(RETRY_MAX_CASE_MS).toBe(ALL_TIERS.CAPTURE_MS + 15_000);
|
||||
for (const row of FILE_RETRY_BUDGETS) {
|
||||
expect(row.retries, row.file).toBe(row.caseMs <= RETRY_MAX_CASE_MS ? (row.file.endsWith('plan-mode-no-op.test.ts') ? 2 : 1) : 0);
|
||||
expect(retriesForFiles([row.file])).toBe(row.retries);
|
||||
}
|
||||
expect(FILE_RETRY_BUDGETS.filter(row => row.retries > 0).map(row => row.file).sort()).toEqual([
|
||||
'test/skill-e2e-auq-matrix.test.ts', 'test/skill-e2e-plan-format.test.ts', 'test/skill-e2e-plan-prosons.test.ts',
|
||||
'test/skill-e2e-qa-callers.test.ts', 'test/skill-llm-eval.test.ts',
|
||||
]);
|
||||
const paid = collectPaidTestFiles();
|
||||
for (const file of SHORT_CASE_RETRY_FILES) {
|
||||
expect(paid, `stale SHORT_CASE_RETRY_FILES entry: ${file}`).toContain(file);
|
||||
expect(FILE_RETRY_BUDGETS.some(row => row.file === file)).toBe(false);
|
||||
const source = read(file);
|
||||
// Declared short budgets only: a JUDGE/CAPTURE tier or a literal at most the
|
||||
// cap, no longer tier and no ms literal past the cap.
|
||||
const literals = [...source.matchAll(/(?<![\w.])(\d{1,3}(?:_\d{3})+|\d{5,})(?![\w.])/g)]
|
||||
.map(match => Number(match[1]!.replace(/_/g, '')));
|
||||
expect(/\b(?:JUDGE_MS|CAPTURE_MS)\b/.test(source) || literals.some(ms => ms >= 60_000 && ms <= RETRY_MAX_CASE_MS), file).toBe(true);
|
||||
expect(source, file).not.toMatch(/\b(?:CAPTURE_LONG_MS|PTY_MS|PTY_LONG_MS|OVERLAY_CASE_[A-Z_]+)\b/);
|
||||
expect(literals.filter(ms => ms > RETRY_MAX_CASE_MS && ms < 10_000_000), file).toEqual([]);
|
||||
expect(retriesForFiles([file])).toBe(1);
|
||||
}
|
||||
for (const file of paid.filter(file => !SHORT_CASE_RETRY_FILES.includes(file) && !FILE_RETRY_BUDGETS.some(row => row.file === file))) {
|
||||
expect(retriesForFiles([file]), file).toBe(0);
|
||||
}
|
||||
expect(retriesForFiles([SHORT_CASE_RETRY_FILES[0]!, 'test/skill-e2e-plan.test.ts'])).toBe(0);
|
||||
});
|
||||
|
||||
test('registration covers exactly the seventeen demonstrated full-file retry gaps', () => {
|
||||
expect(Object.fromEntries(newBudgets.map(row => [row.file, row.shardMs]))).toEqual(expectedWalls);
|
||||
expect(new Set(FILE_RETRY_BUDGETS.map(row => row.file)).size).toBe(19);
|
||||
@@ -86,12 +121,17 @@ test('source allowances retain all captures, cases, and finalization grace', ()
|
||||
]);
|
||||
});
|
||||
|
||||
// Case-sharded files plan one registered case per shard (`<file>#<case id>`).
|
||||
const plannedKey = (file: string) => CASE_SHARDED_FILES.includes(file)
|
||||
? `${file}#${Object.keys(E2E_TOUCHFILES).find(id => E2E_TOUCHFILES[id]!.includes(file))}` : file;
|
||||
|
||||
for (const row of newBudgets) {
|
||||
const key = plannedKey(row.file);
|
||||
const manifest = () => ({ version: 1 as const, tier: 'periodic' as const, evalsAll: false,
|
||||
sliceCount: 1, entries: [{ file: row.file, slice: 1, status: 'planned' as const, budget: resolvePaidShardBudget([row.file]) }] });
|
||||
sliceCount: 1, entries: [{ file: key, slice: 1, status: 'planned' as const, budget: resolvePaidShardBudget([key]) }] });
|
||||
const result = (count = 1) => [{ version: 1 as const, tier: 'periodic' as const, sliceIndex: 1, sliceCount: 1,
|
||||
outcomes: [{ files: [row.file], status: 'passed' as const, exitCode: 0, elapsedMs: 1, executedTests: count,
|
||||
skippedTests: 0, budget: resolvePaidShardBudget([row.file]) }] }];
|
||||
outcomes: [{ files: [key], status: 'passed' as const, exitCode: 0, elapsedMs: 1, executedTests: count,
|
||||
skippedTests: 0, budget: resolvePaidShardBudget([key]) }] }];
|
||||
|
||||
test(`${row.file}: full wall and existing retries propagate through planning`, () => {
|
||||
expect(retriesForFiles([row.file])).toBe(row.retries);
|
||||
@@ -132,8 +172,11 @@ test('fixed AUQ count remains strict while mixed-tier files keep ordinary case h
|
||||
[AUQ_CONSISTENCY_RETRY_BUDGET.file, 0, false],
|
||||
[AUQ_CONSISTENCY_RETRY_BUDGET.file, 1, true],
|
||||
[AUQ_CONSISTENCY_RETRY_BUDGET.file, 2, false],
|
||||
['test/skill-e2e-plan.test.ts', 5, true],
|
||||
['test/skill-e2e-plan.test.ts', 7, true],
|
||||
['test/skill-e2e-ship-docsync.test.ts', 5, true],
|
||||
['test/skill-e2e-ship-docsync.test.ts', 7, true],
|
||||
// A case shard of a case-sharded registered file executes exactly its case.
|
||||
[plannedKey('test/skill-e2e-plan.test.ts'), 1, true],
|
||||
[plannedKey('test/skill-e2e-plan.test.ts'), 2, false],
|
||||
] as const) {
|
||||
const budget = resolvePaidShardBudget([file]);
|
||||
const manifest: any = { version: 1, tier: 'periodic', evalsAll: true, sliceCount: 1,
|
||||
@@ -158,7 +201,7 @@ test('quality judge supervision includes the added judge without changing ordina
|
||||
expect(qualitySource).toContain('const workDeadline = started + JUDGE_MS');
|
||||
expect(qualityBudget.shardMs).toBe((7 * ALL_TIERS.JUDGE_MS + 17 * (ALL_TIERS.JUDGE_MS + 10_000)) * 2 + 120_000);
|
||||
expect(FINDING_RETRY_BUDGETS.map(row => [row.cases, row.testMs, row.retries, row.shardMs])).toEqual([
|
||||
...Array(2).fill([1, 1500000, 1, 3120000]),
|
||||
...Array(2).fill([1, 1500000, 0, 1620000]),
|
||||
]);
|
||||
for (const tier of ['gate', 'periodic'] as const) {
|
||||
const m = buildRunManifest({ tier, sliceCount: 1, evalsAll: true, env: { EVALS_ALL: '1' } });
|
||||
@@ -184,7 +227,7 @@ test('detached PR fallback and release commands cover their actual default worke
|
||||
const prFloor = Math.ceil((Math.ceil(fullGateFiles.length / prWorkers) * 1_800_000 + fullGateFiles.reduce(
|
||||
(total, file) => total + Math.max(0, resolvePaidShardBudget([file]).timeoutMs - 1_800_000), 0,
|
||||
)) / 1000 * 1.05);
|
||||
expect(prFloor).toBe(74_981);
|
||||
expect(prFloor).toBe(71_957);
|
||||
expect(prWall).toBe(92_820_000);
|
||||
expect(prWall).toBeGreaterThanOrEqual(paidShardWallUpperBoundMs(files, prWorkers) + 120_000);
|
||||
|
||||
@@ -203,8 +246,8 @@ test('detached PR fallback and release commands cover their actual default worke
|
||||
)) / 1000 * 1.05));
|
||||
}
|
||||
const detachedReleaseWall = Number(scripts['eval:bg:release'].match(/--timeout (\d+)/)?.[1]) * 1000;
|
||||
expect(releaseFloors).toEqual([42_851, 37_727]);
|
||||
expect(releaseFloors.reduce((total, floor) => total + floor, 0)).toBe(80_578);
|
||||
expect(releaseFloors).toEqual([26_597, 30_797]);
|
||||
expect(releaseFloors.reduce((total, floor) => total + floor, 0)).toBe(57_394);
|
||||
expect(detachedReleaseWall).toBe(116_700_000);
|
||||
expect(detachedReleaseWall).toBeGreaterThanOrEqual(releaseWall + 120_000);
|
||||
});
|
||||
@@ -217,10 +260,10 @@ const cliOptions = (step: { run: string; env?: Record<string, string> }) => {
|
||||
return parseCliOptions(args, step.env ?? {});
|
||||
};
|
||||
|
||||
test('both gate executors cover the complete census without increasing aggregate workers', () => {
|
||||
test('both gate executors plan the complete census and supervise every planned slice', () => {
|
||||
const periodic: any = Bun.YAML.parse(read('.github/workflows/evals-periodic.yml'));
|
||||
const main: any = Bun.YAML.parse(read('.github/workflows/evals.yml'));
|
||||
for (const [workflow, jobName, workers, slices] of [[main, 'eval-slices', 2, 7], [periodic, 'gate-census', 1, 7]] as const) {
|
||||
for (const [workflow, jobName, prefix, skipJudges] of [[main, 'eval-slices', '', false], [periodic, 'gate-census', 'gate_', true]] as const) {
|
||||
const planner = workflow.jobs['plan-slices'];
|
||||
const executor = workflow.jobs[jobName];
|
||||
const emit = planner.steps.filter((step: any) => step.run?.includes('EVALS_TIER=gate ') && step.run.includes('--emit-plan '));
|
||||
@@ -230,41 +273,35 @@ test('both gate executors cover the complete census without increasing aggregate
|
||||
const planned = cliOptions(emit[0]), active = cliOptions(execute[0]);
|
||||
expect(planned.tier).toBe('gate');
|
||||
expect(active.tier).toBe('gate');
|
||||
expect(active.jobs).toBe(workers);
|
||||
expect(active.jobs).toBe(2);
|
||||
expect(planned.jobs).toBe(active.jobs);
|
||||
expect(planned.sliceBudgetMs).toBe(540_000);
|
||||
expect(planned.skipJudges).toBe(skipJudges);
|
||||
expect(execute[0].env.EVALS_CONCURRENCY).toBe('2');
|
||||
expect(executor.strategy['fail-fast']).toBe(false);
|
||||
expect(executor.strategy.matrix.slice).toEqual(Array.from({ length: slices }, (_, i) => i + 1));
|
||||
expect(planned.slices).toBe(slices);
|
||||
const manifest = buildRunManifest({ tier: 'gate', sliceCount: planned.slices, evalsAll: true, env: { EVALS_ALL: '1' } });
|
||||
expect(manifest.entries.filter(row => row.status === 'planned')).toHaveLength(46);
|
||||
const files = manifest.entries.filter(row => row.status === 'planned').map(row => row.file);
|
||||
expect(new Set(files).size).toBe(46);
|
||||
expect(executor.strategy.matrix.slice).toBe(`\${{ fromJSON(needs.plan-slices.outputs.${prefix}slices) }}`);
|
||||
expect(executor['timeout-minutes']).toBe(`\${{ fromJSON(needs.plan-slices.outputs.${prefix}timeout_minutes) }}`);
|
||||
const manifest = buildRunManifest({ tier: 'gate', sliceBudgetMs: planned.sliceBudgetMs!, jobs: planned.jobs, evalsAll: true, env: { EVALS_ALL: '1' }, skipJudges });
|
||||
const files = [...new Set(manifest.entries.filter(row => row.status === 'planned').map(row => shardFile(row.file)))];
|
||||
expect(files).toContain('test/skill-e2e-ship-skip.test.ts');
|
||||
expect(files.sort()).toEqual(selectPaidTestFiles(collectPaidTestFiles(), 'gate').selected.sort());
|
||||
const walls = executor.strategy.matrix.slice.map((slice: number) => paidShardWallUpperBoundMs(
|
||||
manifest.entries.filter(row => row.status === 'planned' && row.slice === slice).map(row => row.file), workers,
|
||||
));
|
||||
expect(executor['timeout-minutes'] * 60_000).toBeGreaterThanOrEqual(Math.max(...walls) + 20 * 60_000);
|
||||
expect(files.sort()).toEqual(selectPaidTestFiles(collectPaidTestFiles(), 'gate').selected
|
||||
.filter(file => !skipJudges || !file.startsWith('test/skill-llm-eval')).sort());
|
||||
const walls = Array.from({ length: manifest.sliceCount }, (_, i) => sliceSupervisedWallMs(sliceExecutionOrder(
|
||||
manifest.entries.filter(row => row.status === 'planned' && row.slice === i + 1)).map(row => row.file), active.jobs));
|
||||
expect(manifest.plan!.ciTimeoutMinutes * 60_000).toBeGreaterThanOrEqual(Math.max(...walls) + 20 * 60_000);
|
||||
expect(manifest.plan!.ciTimeoutMinutes).toBeLessThanOrEqual(360);
|
||||
expect(manifest.sliceCount).toBeLessThanOrEqual(executor.strategy['max-parallel']);
|
||||
if (jobName === 'gate-census') {
|
||||
expect(Math.max(...walls)).toBe(16_440_000);
|
||||
expect(executor['timeout-minutes']).toBe(352);
|
||||
expect(emit[0].env.EVALS_ALL).toBe('1');
|
||||
expect(executor.strategy['max-parallel']).toBe(4);
|
||||
expect(executor.strategy['max-parallel'] * active.jobs).toBe(4);
|
||||
expect(emit[0].run).toContain('--emit-plan /tmp/gate-census-plan/manifest.json');
|
||||
expect(execute[0].run).toContain('--plan /tmp/gate-census-plan/manifest.json --slice ${{ matrix.slice }}');
|
||||
expect(executor.steps.some((step: any) => step.with?.name === 'gate-census-plan')).toBe(true);
|
||||
expect(executor.steps.filter((step: any) => step.run?.includes('--emit-plan '))).toHaveLength(0);
|
||||
} else {
|
||||
expect(Math.max(...walls)).toBe(12_720_000);
|
||||
expect(executor['timeout-minutes']).toBe(265);
|
||||
expect(executor.strategy['max-parallel']).toBe(6);
|
||||
expect(executor.strategy['max-parallel'] * active.jobs).toBe(12);
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
test('the periodic executor supervises every actual case and retry within its CI wall', () => {
|
||||
test('the periodic executor supervises every actual case and retry within its planned CI wall', () => {
|
||||
const workflow: any = Bun.YAML.parse(read('.github/workflows/evals-periodic.yml'));
|
||||
const executor = workflow.jobs['eval-slices'];
|
||||
const emit = workflow.jobs['plan-slices'].steps.filter((step: any) =>
|
||||
@@ -274,21 +311,19 @@ test('the periodic executor supervises every actual case and retry within its CI
|
||||
expect(execute).toHaveLength(1);
|
||||
const planned = cliOptions(emit[0]), active = cliOptions(execute[0]);
|
||||
expect(planned.tier).toBe('periodic');
|
||||
expect(planned.slices).toBe(7);
|
||||
expect(planned.sliceBudgetMs).toBe(540_000);
|
||||
expect(active.jobs).toBe(2);
|
||||
expect(executor.strategy.matrix.slice).toEqual(Array.from({ length: planned.slices }, (_, i) => i + 1));
|
||||
const manifest = buildRunManifest({ tier: 'periodic', sliceCount: planned.slices,
|
||||
expect(planned.jobs).toBe(active.jobs);
|
||||
const manifest = buildRunManifest({ tier: 'periodic', sliceBudgetMs: planned.sliceBudgetMs!, jobs: planned.jobs,
|
||||
evalsAll: true, env: { EVALS_ALL: '1' } });
|
||||
const census = manifest.entries.filter(row => row.status === 'planned');
|
||||
expect(census).toHaveLength(70);
|
||||
expect(new Set(census.map(row => shardFile(row.file)))).toEqual(new Set(selectPaidTestFiles(collectPaidTestFiles(), 'periodic').selected));
|
||||
expect(census.find(row => row.file === 'test/skill-llm-eval.test.ts')?.budget?.timeoutMs).toBe(6_220_000);
|
||||
const walls = executor.strategy.matrix.slice.map((slice: number) => paidShardWallUpperBoundMs(
|
||||
census.filter(row => row.slice === slice).map(row => row.file), active.jobs,
|
||||
));
|
||||
expect(Math.max(...walls)).toBe(14_680_000);
|
||||
expect(executor.strategy['max-parallel']).toBe(8);
|
||||
expect(executor['timeout-minutes']).toBe(360);
|
||||
expect(executor['timeout-minutes'] * 60_000).toBeGreaterThanOrEqual(Math.max(...walls) + 20 * 60_000);
|
||||
const walls = Array.from({ length: manifest.sliceCount }, (_, i) => sliceSupervisedWallMs(sliceExecutionOrder(
|
||||
census.filter(row => row.slice === i + 1)).map(row => row.file), active.jobs));
|
||||
expect(manifest.plan!.ciTimeoutMinutes * 60_000).toBeGreaterThanOrEqual(Math.max(...walls) + 20 * 60_000);
|
||||
expect(manifest.plan!.ciTimeoutMinutes).toBeLessThanOrEqual(360);
|
||||
expect(manifest.sliceCount).toBeLessThanOrEqual(executor.strategy['max-parallel']);
|
||||
});
|
||||
|
||||
test('gate census requires all seven distinct slice results and its own reconciliation', () => {
|
||||
|
||||
+110
-21
@@ -6,8 +6,9 @@
|
||||
* - per-slice selector divergence → ONE planner manifest, executors consume
|
||||
* - hollow lanes → a slice with no artifact is a FAILURE, not an absence
|
||||
* - hollow shards → EVALS_ALL + exit 0 + zero executed tests ≠ pass
|
||||
* - retry parity → the old matrix rows' earned `retries: 2` survive as a
|
||||
* literals map, not folklore
|
||||
* - retry policy → a timed-out attempt is a verdict; only short-case files retry
|
||||
* - budget packing → recorded work packs into ~9-minute executors whose count
|
||||
* and CI job timeout come from the plan
|
||||
*/
|
||||
import { describe, expect, test } from 'bun:test';
|
||||
import * as fs from 'node:fs';
|
||||
@@ -21,13 +22,18 @@ import {
|
||||
buildRunManifest,
|
||||
loadPaidTestDurations,
|
||||
mergePaidTestDurations,
|
||||
packBySliceBudget,
|
||||
estimatedSliceMs,
|
||||
sliceExecutionOrder,
|
||||
sliceSupervisedWallMs,
|
||||
writePaidTestDurations,
|
||||
CI_SETUP_ALLOWANCE_MINUTES,
|
||||
paidShardWallUpperBoundMs,
|
||||
parseCliOptions,
|
||||
parseRunManifest,
|
||||
resolvePaidShardBudget,
|
||||
SUPERVISED_WORKER_COUNTS,
|
||||
retriesForFiles,
|
||||
RETRY_OVERRIDES,
|
||||
summarize,
|
||||
summaryExitCode,
|
||||
verifySliceResults,
|
||||
@@ -46,6 +52,7 @@ const outcome = (over: Partial<ShardOutcome>): ShardOutcome => ({
|
||||
elapsedMs: 1000,
|
||||
groupPid: null,
|
||||
executedTests: 3,
|
||||
skippedTests: null,
|
||||
...over,
|
||||
});
|
||||
|
||||
@@ -142,7 +149,8 @@ describe('recorded-duration slice packing', () => {
|
||||
|
||||
test('the PR-profile file set spreads recorded time instead of stacking it', () => {
|
||||
const env = { EVALS_ALL: '1' };
|
||||
const discovered = Object.keys(recorded);
|
||||
// Seed files only: case-shard keys (`<file>#<case>`) come from expansion.
|
||||
const discovered = Object.keys(recorded).filter(key => !key.includes('#'));
|
||||
const load = (files: string[]) => files.reduce((sum, file) => sum + recorded[file], 0);
|
||||
const packed = lanes(buildRunManifest({ tier: 'gate', sliceCount: 6, evalsAll: true, env, discovered })).map(load);
|
||||
const baseline = lanes(buildRunManifest({ tier: 'gate', sliceCount: 6, evalsAll: true, env, discovered, durations: {} })).map(load);
|
||||
@@ -171,6 +179,88 @@ describe('recorded-duration slice packing', () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe('budget slice packing', () => {
|
||||
const s = (seconds: number) => seconds * 1000;
|
||||
|
||||
test('best-fit packs recorded work under the budget; unknown and over-budget work gets its own runner', () => {
|
||||
const recorded = { 'test/a.test.ts': s(500), 'test/b.test.ts': s(300), 'test/c.test.ts': s(240), 'test/d.test.ts': s(100), 'test/long.test.ts': s(900) };
|
||||
const files = [...Object.keys(recorded), 'test/unknown.test.ts'];
|
||||
const plan = packBySliceBudget(files, s(540), 2, recorded);
|
||||
expect(plan.slices.flat().sort()).toEqual([...files].sort());
|
||||
expect(plan.estimates['test/unknown.test.ts']).toBe(s(540));
|
||||
for (const [index, slice] of plan.slices.entries()) {
|
||||
expect(plan.estimatedSliceMs[index]).toBe(estimatedSliceMs(slice, file => plan.estimates[file]!, 2));
|
||||
if (slice.length > 1) expect(plan.estimatedSliceMs[index]).toBeLessThanOrEqual(s(540));
|
||||
}
|
||||
expect(plan.slices).toContainEqual(['test/long.test.ts']);
|
||||
expect(plan.slices.find(slice => slice.includes('test/unknown.test.ts'))!.length).toBeLessThanOrEqual(2);
|
||||
// Deterministic regardless of discovery order.
|
||||
expect(packBySliceBudget([...files].reverse(), s(540), 2, recorded)).toEqual(plan);
|
||||
const worst = Math.max(...plan.slices.map(slice => sliceSupervisedWallMs(slice, 2)));
|
||||
expect(plan.ciTimeoutMinutes).toBe(Math.ceil(worst / 60_000) + CI_SETUP_ALLOWANCE_MINUTES);
|
||||
expect(packBySliceBudget([], s(540), 2, recorded)).toMatchObject({ slices: [[]], ciTimeoutMinutes: CI_SETUP_ALLOWANCE_MINUTES });
|
||||
});
|
||||
|
||||
test('overlays keep one final slice at their one-at-a-time admission', () => {
|
||||
const overlays = ['test/skill-e2e-overlay-harness-a.test.ts', 'test/skill-e2e-overlay-harness-b.test.ts'];
|
||||
const plan = packBySliceBudget(['test/a.test.ts', ...overlays], s(540), 2, { [overlays[0]!]: s(100), [overlays[1]!]: s(200), 'test/a.test.ts': s(10) });
|
||||
expect(plan.slices.at(-1)).toEqual([overlays[1], overlays[0]]);
|
||||
expect(plan.estimatedSliceMs.at(-1)).toBe(s(300));
|
||||
});
|
||||
|
||||
test('live census plans: multi-file slices stay within the budget and the executor runs longest first', () => {
|
||||
for (const tier of ['gate', 'periodic'] as const) {
|
||||
const manifest = buildRunManifest({ tier, sliceBudgetMs: s(540), jobs: 2, evalsAll: true, env: { EVALS_ALL: '1' } });
|
||||
expect(parseRunManifest(JSON.stringify(manifest))).toEqual(manifest);
|
||||
for (let slice = 1; slice <= manifest.sliceCount; slice++) {
|
||||
const entries = sliceExecutionOrder(manifest.entries.filter(entry => entry.status === 'planned' && entry.slice === slice));
|
||||
const estimate = manifest.plan!.estimatedSliceMs[slice - 1]!;
|
||||
if (entries.length > 1 && !entries.some(entry => entry.file.includes('overlay-harness'))) expect(estimate, `${tier} slice ${slice}`).toBeLessThanOrEqual(s(540));
|
||||
expect(entries.map(entry => entry.estimatedMs)).toEqual([...entries.map(entry => entry.estimatedMs!)].sort((a, b) => b - a));
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
test('plans fail closed on malformed metadata, mixed modes, and a worker count the plan did not supervise', () => {
|
||||
const manifest = buildRunManifest({ tier: 'gate', sliceBudgetMs: s(540), jobs: 2, evalsAll: true, env: { EVALS_ALL: '1' } });
|
||||
for (const broken of [
|
||||
{ ...manifest, plan: { ...manifest.plan!, estimatedSliceMs: [] } },
|
||||
{ ...manifest, plan: { ...manifest.plan!, jobs: 0 } },
|
||||
{ ...manifest, entries: manifest.entries.map(entry => ({ ...entry, estimatedMs: undefined })) },
|
||||
]) expect(() => parseRunManifest(JSON.stringify(broken))).toThrow('slice plan malformed');
|
||||
expect(() => buildRunManifest({ tier: 'gate', sliceCount: 2, sliceBudgetMs: s(540), jobs: 2, evalsAll: true })).toThrow('exactly one');
|
||||
expect(() => buildRunManifest({ tier: 'gate', sliceBudgetMs: s(540), evalsAll: true })).toThrow('explicit positive --jobs');
|
||||
expect(() => parseCliOptions(['--emit-plan', 'x', '--slice-budget', '540'], {})).toThrow('explicit --jobs');
|
||||
expect(() => parseCliOptions(['--emit-plan', 'x', '--slice-budget', '540', '--jobs', '2', '--slices', '3'], {})).toThrow('exactly one');
|
||||
expect(parseCliOptions(['--emit-plan', 'x', '--slice-budget', '540'], { EVALS_JOBS: '2' }).sliceBudgetMs).toBe(s(540));
|
||||
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'paid-plan-jobs-'));
|
||||
try {
|
||||
const planPath = path.join(dir, 'manifest.json');
|
||||
fs.writeFileSync(planPath, JSON.stringify(manifest));
|
||||
const result = spawnSync(process.execPath, [path.join(ROOT, 'scripts/test-paid-shards.ts'), '--plan', planPath, '--slice', '1', '--list', '--jobs', '1'],
|
||||
{ cwd: ROOT, encoding: 'utf8', timeout: 10_000, env: { PATH: path.dirname(process.execPath), HOME: dir, EVALS_TIER: 'gate' } });
|
||||
expect(result.status).toBe(1);
|
||||
expect(result.stderr).toContain('manifest was packed for 2 worker(s) per slice');
|
||||
} finally { fs.rmSync(dir, { recursive: true, force: true }); }
|
||||
});
|
||||
|
||||
test('the duration seed is per tier and a report rewrite keeps the other tiers', () => {
|
||||
const gate = loadPaidTestDurations(ROOT, 'gate');
|
||||
const periodic = loadPaidTestDurations(ROOT, 'periodic');
|
||||
expect(gate['test/skill-e2e-plan.test.ts']).not.toBe(periodic['test/skill-e2e-plan.test.ts']);
|
||||
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'paid-durations-'));
|
||||
try {
|
||||
fs.mkdirSync(path.join(dir, 'scripts'));
|
||||
writePaidTestDurations('periodic', { 'test/p.test.ts': 2_000 }, dir);
|
||||
writePaidTestDurations('gate', { 'test/g.test.ts': 3_000 }, dir);
|
||||
expect(loadPaidTestDurations(dir, 'gate')).toEqual({ 'test/g.test.ts': 3_000 });
|
||||
expect(loadPaidTestDurations(dir, 'periodic')).toEqual({ 'test/p.test.ts': 2_000 });
|
||||
fs.writeFileSync(path.join(dir, 'scripts/paid-test-durations.json'), JSON.stringify({ version: 1, durations: { 'test/g.test.ts': 1 } }));
|
||||
expect(loadPaidTestDurations(dir, 'gate')).toEqual({});
|
||||
} finally { fs.rmSync(dir, { recursive: true, force: true }); }
|
||||
});
|
||||
});
|
||||
|
||||
describe('manifest executor scope', () => {
|
||||
test('list-only validates and prints the selected manifest slice without launching tests or writing results', () => {
|
||||
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'paid-manifest-list-'));
|
||||
@@ -184,8 +274,8 @@ test('local launch sentinel', () => writeFileSync(${JSON.stringify(receipt)}, 't
|
||||
version: 1, tier: 'gate', evalsAll: true, sliceCount: 3, selectionReason: 'local list-only fixture',
|
||||
entries: [
|
||||
{ file, slice: 1, status: 'planned' },
|
||||
{ file: 'test/skill-e2e-plan.test.ts', slice: 2, status: 'planned',
|
||||
budget: resolvePaidShardBudget(['test/skill-e2e-plan.test.ts']) },
|
||||
{ file: 'test/skill-e2e-plan.test.ts#plan-ceo-review', slice: 2, status: 'planned',
|
||||
budget: resolvePaidShardBudget(['test/skill-e2e-plan.test.ts#plan-ceo-review']) },
|
||||
],
|
||||
};
|
||||
const manifestPath = path.join(dir, 'manifest.json');
|
||||
@@ -327,7 +417,7 @@ describe('slice-result reconciliation (report)', () => {
|
||||
tier: 'gate',
|
||||
sliceIndex: index,
|
||||
sliceCount: 2,
|
||||
outcomes: files.map((f) => ({ files: [f], status, exitCode: 0, elapsedMs: 5, executedTests: 2 })),
|
||||
outcomes: files.map((f) => ({ files: [f], status, exitCode: 0, elapsedMs: 5, executedTests: 2, skippedTests: 0 })),
|
||||
});
|
||||
|
||||
test('all slices present and passing → ok', () => {
|
||||
@@ -397,25 +487,24 @@ describe('hollow-shard guard', () => {
|
||||
});
|
||||
|
||||
describe('retry parity', () => {
|
||||
test('registered native workflows preserve main retry policy while overlay attempts stay isolated', () => {
|
||||
test('registered native workflows follow the retry rule while overlay attempts stay isolated', () => {
|
||||
// A 25-minute case is past RETRY_MAX_CASE_MS: its timed-out attempt is the verdict.
|
||||
const native = 'test/skill-e2e-plan-ceo-split-overflow.test.ts';
|
||||
expect(retriesForFiles([native])).toBe(1);
|
||||
expect(retriesForFiles([native.replaceAll('/', '\\')])).toBe(1);
|
||||
expect(buildPaidShardArgs([native], 1_800_000, 2, retriesForFiles([native])).join(' ')).toContain('--retry 1');
|
||||
expect(retriesForFiles([native])).toBe(0);
|
||||
expect(retriesForFiles([native.replaceAll('/', '\\')])).toBe(0);
|
||||
expect(buildPaidShardArgs([native], 1_800_000, 2, retriesForFiles([native])).join(' ')).toContain('--retry 0');
|
||||
const overlay = 'test/skill-e2e-overlay-harness-claude-dedicated-tools-vs-bash.test.ts';
|
||||
expect(retriesForFiles([overlay])).toBe(0);
|
||||
});
|
||||
test('overrides exist only for the files whose matrix rows earned them, and each names a real file', () => {
|
||||
expect(Object.keys(RETRY_OVERRIDES).sort()).toEqual([
|
||||
'test/skill-e2e-office-hours-auto-mode.test.ts',
|
||||
'test/skill-e2e-plan-mode-no-op.test.ts',
|
||||
'test/skill-e2e-workflow.test.ts',
|
||||
]);
|
||||
for (const file of Object.keys(RETRY_OVERRIDES)) {
|
||||
expect(fs.existsSync(path.join(ROOT, file)), `stale RETRY_OVERRIDES entry: ${file}`).toBe(true);
|
||||
test('the matrix-era earned retries now follow the timeout-is-a-verdict rule, and each names a real file', () => {
|
||||
// These three old matrix rows earned `retries: 2`; every one has a
|
||||
// CAPTURE_LONG case, so a timed-out attempt is now their verdict.
|
||||
for (const file of ['test/skill-e2e-office-hours-auto-mode.test.ts', 'test/skill-e2e-plan-mode-no-op.test.ts', 'test/skill-e2e-workflow.test.ts']) {
|
||||
expect(fs.existsSync(path.join(ROOT, file)), `stale retry parity entry: ${file}`).toBe(true);
|
||||
expect(retriesForFiles([file])).toBe(0);
|
||||
}
|
||||
expect(retriesForFiles(['test/skill-e2e-workflow.test.ts'])).toBe(2);
|
||||
expect(retriesForFiles(['test/skill-e2e-retro.test.ts'])).toBe(1);
|
||||
expect(retriesForFiles(['test/skill-e2e-retro.test.ts'])).toBe(0);
|
||||
expect(retriesForFiles(['test/skill-e2e-review.test.ts'])).toBe(1);
|
||||
expect(buildPaidShardArgs(['x'], 1000, 4, 2)).toContain('2');
|
||||
expect(buildPaidShardArgs(['x'], 1000, 4).join(' ')).toContain('--retry 1');
|
||||
});
|
||||
|
||||
+114
-1
@@ -13,6 +13,7 @@ import { describe, test, expect } from 'bun:test';
|
||||
import * as fs from 'fs';
|
||||
import * as os from 'os';
|
||||
import * as path from 'path';
|
||||
import { E2E_TIERS, E2E_TOUCHFILES } from './helpers/touchfiles';
|
||||
|
||||
const ROOT = path.resolve(import.meta.dir, '..');
|
||||
import {
|
||||
@@ -31,7 +32,21 @@ import {
|
||||
summarize,
|
||||
summaryExitCode,
|
||||
tierSkipReason,
|
||||
marathonSkipReason,
|
||||
CASE_SHARDED_FILES,
|
||||
CASE_TEST_NAMES,
|
||||
caseTestNamePattern,
|
||||
expandCaseShards,
|
||||
fileCaseRegistration,
|
||||
resolvePaidShardBudget,
|
||||
retriesForFiles,
|
||||
shardCaseId,
|
||||
shardFile,
|
||||
shardSlug,
|
||||
verifySliceResults,
|
||||
selectPaidTestFiles,
|
||||
buildRunManifest,
|
||||
parseRunManifest,
|
||||
type ShardOutcome,
|
||||
} from '../scripts/test-paid-shards';
|
||||
|
||||
@@ -117,7 +132,105 @@ describe('tier lane skip (B5)', () => {
|
||||
}
|
||||
const workflow = fs.readFileSync(path.join(ROOT, '.github/workflows/evals-periodic.yml'), 'utf8');
|
||||
expect(workflow.match(/--skip-judges/g)).toHaveLength(1);
|
||||
expect(workflow).toMatch(/--tier gate --emit-plan \/tmp\/gate-census-plan\/manifest\.json --slices 7 --skip-judges/);
|
||||
expect(workflow).toMatch(/--tier gate --emit-plan \/tmp\/gate-census-plan\/manifest\.json --slice-budget 540 --jobs 2 --skip-judges/);
|
||||
});
|
||||
});
|
||||
|
||||
describe('marathon tier lane', () => {
|
||||
const file = 'test/skill-e2e-sample.test.ts';
|
||||
const reg = { 'sample-gate': [file], 'sample-long': [file] } as Record<string, string[]>;
|
||||
const tiers = { 'sample-gate': 'gate', 'sample-long': 'marathon' };
|
||||
|
||||
test('a marathon-declared file never enters the gate or periodic lane', () => {
|
||||
const source = "const describeE2E = describeE2ETier('marathon');";
|
||||
expect(classifyPaidTestFile(source, 'marathon')).toEqual({ included: true, reason: "declares tier 'marathon'" });
|
||||
for (const tier of ['gate', 'periodic'] as const) {
|
||||
expect(classifyPaidTestFile(source, tier)).toEqual({ included: false, reason: "declares tier 'marathon' only" });
|
||||
}
|
||||
expect(marathonSkipReason(file, source, {}, {})).toBeNull();
|
||||
});
|
||||
|
||||
test('marathon selects positively: only declared files or files registering a marathon case', () => {
|
||||
expect(marathonSkipReason(file, "testIfSelected('sample-long', async () => {});", reg, tiers)).toBeNull();
|
||||
expect(marathonSkipReason(file, "testIfSelected(name, async () => {});", reg, tiers)).toBeNull();
|
||||
const gateOnly = { 'sample-gate': [file] };
|
||||
for (const source of ["testIfSelected('sample-gate', async () => {});", "testIfSelected(name, async () => {});", ''])
|
||||
expect(marathonSkipReason(file, source, gateOnly, tiers)).toBe('skipped: declares no marathon tier and registers no marathon case');
|
||||
const periodic = "const describeE2E = describeE2ETier('periodic');";
|
||||
expect(classifyPaidTestFile(periodic, 'marathon')).toEqual({ included: false, reason: "declares tier 'periodic' only" });
|
||||
});
|
||||
|
||||
test('a registered marathon case keeps its gate sibling scheduled in the gate lane', () => {
|
||||
const source = "testIfSelected('sample-gate', async () => {}); testIfSelected('sample-long', async () => {});";
|
||||
expect(tierSkipReason(file, source, 'gate', reg, tiers)).toBeNull();
|
||||
expect(tierSkipReason(file, source, 'periodic', reg, tiers)).toBe('skipped: no E2E_TIERS id has tier periodic');
|
||||
});
|
||||
|
||||
test('the live marathon lane only plans files that carry marathon work, never the LLM judges', () => {
|
||||
const { selected, excluded } = selectPaidTestFiles(collectPaidTestFiles(), 'marathon');
|
||||
expect(selected).not.toContain('test/skill-llm-eval.test.ts');
|
||||
for (const file of selected) {
|
||||
const source = fs.readFileSync(path.join(ROOT, file), 'utf8');
|
||||
expect(marathonSkipReason(file, source), file).toBeNull();
|
||||
}
|
||||
expect(selected.length + excluded.length).toBe(collectPaidTestFiles().length);
|
||||
});
|
||||
});
|
||||
|
||||
describe('case-sharded files', () => {
|
||||
// Paid cases only: `if (!evalsEnabled) test(...)` blocks are free checks.
|
||||
const caseNames = (source: string) => [...source.matchAll(
|
||||
/(?<![.\w])(?<!if \(!evalsEnabled\) )(?:testConcurrentIfSelected|testIfSelected|test(?:\.serial|\.concurrent)?)\(\s*(['"])(.+?)\1/g,
|
||||
)].map(match => match[2]!);
|
||||
|
||||
for (const file of CASE_SHARDED_FILES) {
|
||||
test(`${file}: every Bun case is a registered E2E case, so every case gets a shard`, () => {
|
||||
const source = fs.readFileSync(path.join(ROOT, file), 'utf8');
|
||||
const { registered, known } = fileCaseRegistration(file, source);
|
||||
expect(known).toBe(true);
|
||||
expect(caseNames(source).sort()).toEqual(registered.map(id => CASE_TEST_NAMES[id] ?? id).sort());
|
||||
const keys = (['gate', 'periodic', 'marathon'] as const).flatMap(tier => expandCaseShards([file], tier));
|
||||
expect(keys.map(key => shardCaseId(key)).sort()).toEqual([...registered].sort());
|
||||
for (const key of keys) expect(shardFile(key)).toBe(file);
|
||||
});
|
||||
}
|
||||
|
||||
test('a case key runs exactly its case: exact name pattern, own eval slug, per-case supervision', () => {
|
||||
const pattern = new RegExp(caseTestNamePattern(['design-review-detector-shim']));
|
||||
expect(pattern.test('Design review detector shim E2E design-review-detector-shim')).toBe(true);
|
||||
expect(pattern.test('Design review detector shim E2E design-review-detector-shim-dom')).toBe(false);
|
||||
expect(new RegExp(caseTestNamePattern(['plan-review-report'])).test('Plan Review Report E2E /plan-eng-review writes GSTACK REVIEW REPORT to plan file')).toBe(true);
|
||||
const key = 'test/skill-e2e-plan.test.ts#plan-ceo-review';
|
||||
expect(shardSlug([key])).toBe('skill-e2e-plan--plan-ceo-review');
|
||||
expect(shardSlug([key])).not.toBe(shardSlug(['test/skill-e2e-plan.test.ts#plan-eng-review']));
|
||||
expect(retriesForFiles([key])).toBe(retriesForFiles(['test/skill-e2e-plan.test.ts']));
|
||||
const whole = resolvePaidShardBudget(['test/skill-e2e-plan.test.ts']);
|
||||
const one = resolvePaidShardBudget([key]);
|
||||
expect(one.policyId).toBe(whole.policyId);
|
||||
expect(one.timeoutMs).toBeLessThan(whole.timeoutMs);
|
||||
expect(planPaidShards([key, 'test/skill-e2e-plan.test.ts#plan-eng-review', 'test/a.test.ts'], { maxFilesPerShard: 3 }))
|
||||
.toEqual([['test/a.test.ts'], [key], ['test/skill-e2e-plan.test.ts#plan-eng-review']]);
|
||||
});
|
||||
|
||||
test('manifests plan each case once and results must execute exactly that case', () => {
|
||||
const manifest = buildRunManifest({ tier: 'gate', sliceBudgetMs: 540_000, jobs: 2, evalsAll: true, env: { EVALS_ALL: '1' } });
|
||||
const planned = manifest.entries.filter(entry => entry.status === 'planned');
|
||||
for (const file of CASE_SHARDED_FILES) {
|
||||
expect(planned.some(entry => entry.file === file)).toBe(false);
|
||||
const gateIds = Object.keys(E2E_TOUCHFILES).filter(id => E2E_TOUCHFILES[id]!.includes(file) && E2E_TIERS[id] === 'gate');
|
||||
expect(planned.filter(entry => shardFile(entry.file) === file).map(entry => shardCaseId(entry.file)).sort()).toEqual(gateIds.sort());
|
||||
}
|
||||
const results = Array.from({ length: manifest.sliceCount }, (_, i) => ({ version: 1 as const, tier: 'gate' as const, sliceIndex: i + 1, sliceCount: manifest.sliceCount,
|
||||
outcomes: planned.filter(entry => entry.slice === i + 1).map(entry => ({ files: [entry.file], status: 'passed' as const, exitCode: 0, elapsedMs: 1,
|
||||
executedTests: shardCaseId(entry.file) ? 3 : 1, skippedTests: shardCaseId(entry.file) ? 2 : 0, ...(entry.budget ? { budget: entry.budget } : {}) })) }));
|
||||
expect(verifySliceResults(manifest, results).problems.filter(problem => problem.includes('Case shard'))).toEqual([]);
|
||||
const empty = structuredClone(results);
|
||||
const victim = empty.flatMap(result => result.outcomes).find(outcome => shardCaseId(outcome.files[0]!))!;
|
||||
victim.skippedTests = victim.executedTests;
|
||||
expect(verifySliceResults(manifest, empty).problems).toContain(`Case shard must execute exactly its one case: ${victim.files[0]}`);
|
||||
const whole: any = structuredClone(manifest);
|
||||
whole.entries.push({ file: CASE_SHARDED_FILES[0], slice: 1, status: 'planned', estimatedMs: 1 });
|
||||
expect(() => parseRunManifest(JSON.stringify(whole))).toThrow('one registered case per shard');
|
||||
});
|
||||
});
|
||||
|
||||
|
||||
Reference in new issue
Block a user