From 3c441fe311cf99c7507ca676e5740aa6cf109534 Mon Sep 17 00:00:00 2001 From: garrytan Date: Tue, 29 Sep 2026 18:56:27 +0000 Subject: [PATCH 01/15] test(evals): add E2E_KINDS, BEHAVIOR_WHY, EVAL_POLICY and CASE_QUARANTINE skeletons Every E2E_TIERS and LLM_JUDGE_TOUCHFILES key starts as 'rule'; BEHAVIOR_WHY and CASE_QUARANTINE start empty. EVAL_POLICY pre-registers the approved panel (3, majority 2), quarantine entry 0.95/10 and exit 0.97/10, 10% cap, 8-weekly-run expiry, Fisher drift alarm and one INFRA re-dispatch. --- test/helpers/periodic-exclude-data.ts | 54 ++++++ test/helpers/touchfiles-data.ts | 258 ++++++++++++++++++++++++++ 2 files changed, 312 insertions(+) diff --git a/test/helpers/periodic-exclude-data.ts b/test/helpers/periodic-exclude-data.ts index 7538b4ccc..5959f44d7 100644 --- a/test/helpers/periodic-exclude-data.ts +++ b/test/helpers/periodic-exclude-data.ts @@ -59,3 +59,57 @@ export const CASE_CI_EXCLUDE: Record= k passing trials with + * no contract violation. Rule and judge cases run one trial. + * quarantine - entry below `entry.rate` per trial over >= `entry.minTrials` + * new-policy trials; exit at >= `exit.rate` over >= + * `exit.minTrials`; at most `capFraction` of each tier's + * blocking cases; an entry expires after `expiryWeeklyRuns`. + * drift - one-sided Fisher exact alarm between input-identity series. + * infraRedispatch - a census whose every red verdict is machine-classified + * INFRA or INCOMPLETE may be re-dispatched this many times as + * a new run; both runs are reported. + */ +export const EVAL_POLICY = { + version: 1, + panel: { n: 3, k: 2 }, + quarantine: { + entry: { rate: 0.95, minTrials: 10 }, + exit: { rate: 0.97, minTrials: 10 }, + capFraction: 0.10, + expiryWeeklyRuns: 8, + }, + drift: { fisherAlpha: 0.05, fisherMinPerSide: 6 }, + infraRedispatch: 1, +} as const; + +/** + * Quarantined paid cases, keyed by registry id (an E2E_TIERS key). A + * quarantined case still runs its full panel and reports in every lane, but + * its failed verdict cannot fail the lane unless the panel is a hard break + * (0 of n) or a trial violated a contract; it never counts as passing + * coverage. An entry needs the entry rule met on the current input identity, + * a written diagnosis that the failures are detector, harness or model-latency + * failures (a product defect is never quarantined), and unchanged case + * touchfiles in the change that adds it. Pinned by + * test/periodic-exclude-policy.test.ts. + * reason - the written diagnosis + * tracking - issue or TODOS pointer + * owner - who removes it + * enteredAt - ISO date the entry landed (expiry counts weekly runs from here) + * exit - the measurable exit condition + */ +export const CASE_QUARANTINE: Record = {}; diff --git a/test/helpers/touchfiles-data.ts b/test/helpers/touchfiles-data.ts index 5d9711d2c..18fc17f77 100644 --- a/test/helpers/touchfiles-data.ts +++ b/test/helpers/touchfiles-data.ts @@ -1573,3 +1573,261 @@ export const GLOBAL_TOUCHFILES = [ // diffed per key, so a data-only edit runs just the affected tests. // Map-diff fails CLOSED — any error on that path still runs everything. ]; + +/** + * Eval kind per live case (every E2E_TIERS and LLM_JUDGE_TOUCHFILES key). + * The kind fixes the trial policy before the run (EVAL_POLICY in + * periodic-exclude-data.ts): + * rule - one trial; any failed assertion fails the verdict. The default. + * behavior - a panel of independent trials, PASS at the policy majority; + * needs a BEHAVIOR_WHY entry naming the tolerated deviation. + * judge - an LLM-judge score of a static input, sampled as a panel. + * Reclassification is a reviewed diff, never a runtime switch. + */ +export const E2E_KINDS: Record = { + 'ship-skipped-queued-finding': 'rule', + 'investigate-owned-completion': 'rule', + 'investigate-owned-abort': 'rule', + 'investigate-owned-ending-error': 'rule', + 'shared-libs-review-path-eligibility': 'rule', + 'shared-libs-review-index-flags': 'rule', + 'shared-libs-review-prior-coverage': 'rule', + 'shared-libs-codex-read-only': 'rule', + 'shared-libs-read-only': 'rule', + 'shared-libs-unsupported-git': 'rule', + 'shared-libs-review-lifecycle': 'rule', + 'shared-libs-review-revalidation': 'rule', + 'shared-libs-opportunity-judgment': 'rule', + 'shared-libs-pr-coverage': 'rule', + 'shared-libs-plan-callers': 'rule', + 'browse-basic': 'rule', + 'browse-snapshot': 'rule', + 'aside-browse-basic': 'rule', + 'aside-browse-flow': 'rule', + 'aside-qa-quick': 'rule', + 'aside-scrape-json': 'rule', + 'aside-canary-quick': 'rule', + 'hermetic-canary': 'rule', + 'hermetic-sentinel': 'rule', + 'skillmd-setup-discovery': 'rule', + 'skillmd-no-local-binary': 'rule', + 'skillmd-outside-git': 'rule', + 'session-awareness': 'rule', + 'operational-learning': 'rule', + 'first-task-scaffold': 'rule', + 'qa-quick': 'rule', + 'qa-b6-static': 'rule', + 'qa-b7-spa': 'rule', + 'qa-b8-checkout': 'rule', + 'qa-only-no-fix': 'rule', + 'qa-fix-loop': 'rule', + 'qa-bootstrap': 'rule', + 'review-exploratory-small-cli': 'rule', + 'ship-exploratory-small-cli': 'rule', + 'ship-exploratory-unavailable': 'rule', + 'ship-exploratory-plan-checks': 'rule', + 'ship-exploratory-late-input': 'rule', + 'qa-functional-cli-report': 'rule', + 'qa-functional-webhook-report': 'rule', + 'qa-functional-cli-fix': 'rule', + 'qa-functional-webhook-fix': 'rule', + 'review-sql-injection': 'rule', + 'review-enum-completeness': 'rule', + 'review-base-branch': 'rule', + 'review-design-lite': 'rule', + 'review-coverage-audit': 'rule', + 'review-dashboard-via': 'rule', + 'review-army-migration-safety': 'rule', + 'review-army-perf-n-plus-one': 'rule', + 'review-army-delivery-audit': 'rule', + 'review-army-quality-score': 'rule', + 'review-army-json-findings': 'rule', + 'review-army-red-team': 'rule', + 'review-army-consensus': 'rule', + 'review-army-simplification': 'rule', + 'review-army-simplification-precision': 'rule', + 'office-hours-spec-review': 'rule', + 'office-hours-brain-writeback': 'rule', + 'gbrain-roundtrip-local': 'rule', + 'sync-gbrain-read-ready': 'rule', + 'sync-gbrain-read-unknown': 'rule', + 'office-hours-forcing-energy': 'rule', + 'office-hours-builder-wildness': 'rule', + 'plan-ceo-review': 'rule', + 'plan-ceo-review-selective': 'rule', + 'plan-ceo-review-benefits': 'rule', + 'plan-ceo-review-expansion-energy': 'rule', + 'plan-eng-review': 'rule', + 'plan-eng-review-artifact': 'rule', + 'plan-eng-coverage-audit': 'rule', + 'plan-review-report': 'rule', + 'plan-ceo-review-plan-mode': 'rule', + 'plan-eng-review-plan-mode': 'rule', + 'plan-design-review-plan-mode': 'rule', + 'plan-devex-review-plan-mode': 'rule', + 'plan-mode-no-op': 'rule', + 'office-hours-auto-mode': 'rule', + 'auto-decide-preserved': 'rule', + 'auq-format-gate': 'rule', + 'plan-ceo-mode-routing': 'rule', + 'plan-design-with-ui-scope': 'rule', + 'tpa-present': 'rule', + 'tpa-absent-linux': 'rule', + 'tpa-broken': 'rule', + 'tpa-absent-darwin': 'rule', + 'tpa-apple-ban': 'rule', + 'ship-section-loading': 'rule', + 'plan-ceo-section-loading': 'rule', + 'carve-section-loading': 'rule', + 'plan-eng-finding-floor': 'rule', + 'plan-ceo-finding-floor': 'rule', + 'plan-design-finding-floor': 'rule', + 'plan-devex-finding-floor': 'rule', + 'plan-eng-multi-finding-batching': 'rule', + 'plan-ceo-split-overflow': 'rule', + 'setup-gbrain-remote': 'rule', + 'setup-gbrain-bad-token': 'rule', + 'setup-gbrain-path4-local-pglite': 'rule', + 'plan-ceo-review-format-mode': 'rule', + 'plan-ceo-review-format-approach': 'rule', + 'plan-eng-review-format-coverage': 'rule', + 'plan-eng-review-format-kind': 'rule', + 'office-hours-phase4-fork': 'rule', + 'llm-judge-recommendation': 'rule', + 'plan-ceo-review-prosons-cadence': 'rule', + 'plan-review-prosons-format': 'rule', + 'plan-review-prosons-hardstop-neg': 'rule', + 'plan-review-prosons-neutral-neg': 'rule', + 'plan-tune-inspect': 'rule', + 'codex-offered-office-hours': 'rule', + 'codex-offered-ceo-review': 'rule', + 'codex-offered-design-review': 'rule', + 'codex-offered-eng-review': 'rule', + 'timeline-event-flow': 'rule', + 'context-recovery-artifacts': 'rule', + 'context-save-writes-file': 'rule', + 'context-restore-loads-latest': 'rule', + 'context-save-routing': 'rule', + 'context-save-then-restore-roundtrip': 'rule', + 'context-restore-fragment-match': 'rule', + 'context-restore-empty-state': 'rule', + 'context-restore-list-delegates': 'rule', + 'context-restore-legacy-compat': 'rule', + 'context-save-list-current-branch': 'rule', + 'context-save-list-all-branches': 'rule', + 'ship-base-branch': 'rule', + 'ship-local-workflow': 'rule', + 'ship-managed-hook-refresh': 'rule', + 'ship-unmanaged-hook-consent': 'rule', + 'ship-local-hook-preservation': 'rule', + 'ship-coverage-audit': 'rule', + 'ship-triage': 'rule', + 'ship-docsync-missing-marker': 'rule', + 'ship-docsync-missing-asset': 'rule', + 'ship-docsync-launch-failure': 'rule', + 'ship-docsync-timeout-unsettled': 'rule', + 'ship-docsync-late-result': 'rule', + 'ship-docsync-stale-before': 'rule', + 'ship-docsync-stale-after': 'rule', + 'ship-docsync-recovery': 'rule', + 'ship-docsync-completion': 'rule', + 'ship-docsync-current': 'rule', + 'ship-docsync-failure': 'rule', + 'ship-docsync-store': 'rule', + 'docsync-spawned': 'rule', + 'retro': 'rule', + 'retro-base-branch': 'rule', + 'cso-full-audit': 'rule', + 'cso-diff-mode': 'rule', + 'cso-infra-scope': 'rule', + 'learnings-show': 'rule', + 'document-release': 'rule', + 'codex-review': 'rule', + 'codex-discover-skill': 'rule', + 'codex-review-findings': 'rule', + 'outside-voice-codex-to-claude-code': 'rule', + 'outside-voice-claude-code-to-codex': 'rule', + 'outside-plan-disabled-no-fallback': 'rule', + 'codex-sol-scope-termination': 'rule', + 'design-consultation-core': 'rule', + 'design-consultation-existing': 'rule', + 'design-consultation-research': 'rule', + 'design-consultation-preview': 'rule', + 'plan-design-review-no-ui-scope': 'rule', + 'design-review-fix': 'rule', + 'design-review-detector-shim': 'rule', + 'design-review-detector-shim-dom': 'rule', + 'design-review-plugin-handoff': 'rule', + 'design-html-slop-gate': 'rule', + 'diagram-triplet': 'rule', + 'diagram-authoring-quality': 'rule', + 'gstack-upgrade-happy-path': 'rule', + 'land-and-deploy-workflow': 'rule', + 'land-and-deploy-first-run': 'rule', + 'land-and-deploy-review-gate': 'rule', + 'canary-workflow': 'rule', + 'benchmark-workflow': 'rule', + 'setup-deploy-workflow': 'rule', + 'autoplan-dual-voice': 'rule', + 'benchmark-providers-live': 'rule', + 'scrape-match-path': 'rule', + 'scrape-prototype-path': 'rule', + 'skillify-happy-path': 'rule', + 'skillify-provenance-refusal': 'rule', + 'skillify-approval-reject': 'rule', + 'journey-ideation': 'rule', + 'journey-plan-eng': 'rule', + 'journey-debug': 'rule', + 'journey-qa': 'rule', + 'journey-code-review': 'rule', + 'journey-ship': 'rule', + 'journey-docs': 'rule', + 'journey-retro': 'rule', + 'journey-design-system': 'rule', + 'journey-visual-qa': 'rule', + 'ios-qa-device': 'rule', + 'arm-benchmark-native-overbuild': 'rule', + 'arm-benchmark-crud-endpoint': 'rule', + 'arm-benchmark-bugfix-decoys': 'rule', + 'office-hours-section-loading': 'rule', + 'office-hours-design-draft': 'rule', + 'plan-decision-classification': 'rule', + 'plan-devex-peer-comparison-classification': 'rule', + 'health-reporting': 'rule', + 'overlay-harness-claude-dedicated-tools-vs-bash': 'rule', + 'overlay-harness-opus-4-7-effort-match-trivial': 'rule', + 'overlay-harness-opus-4-7-literal-interpretation': 'rule', + 'overlay-harness-claude-dedicated-tools-vs-bash-sonnet': 'rule', + 'journey-negatives': 'rule', + 'review/SKILL.md workflow': 'rule', + 'setup-browser-cookies/SKILL.md workflow': 'rule', + 'browse/SKILL.md reference': 'rule', + 'setup block': 'rule', + 'qa/SKILL.md workflow': 'rule', + 'qa/SKILL.md health rubric': 'rule', + 'qa/SKILL.md anti-refusal': 'rule', + 'cross-skill greptile consistency': 'rule', + 'ship/SKILL.md workflow': 'rule', + 'document-release/SKILL.md workflow': 'rule', + 'plan-ceo-review/SKILL.md modes': 'rule', + 'plan-eng-review/SKILL.md sections': 'rule', + 'plan-design-review/SKILL.md passes': 'rule', + 'design-review/SKILL.md fix loop': 'rule', + 'design-consultation/SKILL.md research': 'rule', + 'land-and-deploy/SKILL.md workflow': 'rule', + 'canary/SKILL.md monitoring loop': 'rule', + 'benchmark/SKILL.md perf collection': 'rule', + 'setup-deploy/SKILL.md platform setup': 'rule', + 'retro/SKILL.md instructions': 'rule', + 'qa-only/SKILL.md workflow': 'rule', + 'gstack-upgrade/SKILL.md upgrade flow': 'rule', + 'sync-gbrain/SKILL.md read-only readiness': 'rule', + 'voice directive tone': 'rule', +}; + +/** + * One-line tolerance for every behavior-kind case: why an occasional + * deviation is acceptable product behavior. Keys equal the behavior ids of + * E2E_KINDS; values are non-empty. + */ +export const BEHAVIOR_WHY: Record = {}; From adced7e046121ff3cd0efd146cc7e963c774848b Mon Sep 17 00:00:00 2001 From: garrytan Date: Tue, 29 Sep 2026 18:56:27 +0000 Subject: [PATCH 02/15] test(evals): add trial records, panelVerdict, expectContract and trial-outcomes JSONL EvalTestEntry gains case_id, kind, trial, panel, failure_class and policy_version, stamped from the runner's TRIAL_ENV on isolated trial shards. panelVerdict() is the single verdict function (INCOMPLETE on missing or duplicate trials, contract veto at any count, quarantine hard-break rule, INFRA/INCOMPLETE machine classification). expectContract() records failure_class 'contract' on the collector entry and a sidecar before throwing. trial-outcomes JSONL has a fail-closed writer and a data-only reader. --- scripts/typecheck-test-baseline.json | 1 - test/helpers/eval-store.test.ts | 202 ++++++++++++++- test/helpers/eval-store.ts | 365 ++++++++++++++++++++++++++- test/setup-gbrain-fixture.test.ts | 6 +- 4 files changed, 568 insertions(+), 6 deletions(-) diff --git a/scripts/typecheck-test-baseline.json b/scripts/typecheck-test-baseline.json index d25bf86d5..c309682d1 100644 --- a/scripts/typecheck-test-baseline.json +++ b/scripts/typecheck-test-baseline.json @@ -448,7 +448,6 @@ "test/section-capture-native-tools.test.ts\tTS2339\tProperty 'CI' does not exist on type '{ NODE_ENV?: string | undefined; TZ?: string | undefined; PATH: string; }'.": 1, "test/session-runner-browse-errors.test.ts\tTS2769\tNo overload matches this call. The last overload gave the following error. Argument of type 'never[]' is not assignable to parameter of type 'undefined'.": 1, "test/session-runner-browse-errors.test.ts\tTS7006\tParameter 'row' implicitly has an 'any' type.": 6, - "test/setup-gbrain-fixture.test.ts\tTS2352\tConversion of type '{ addTest: (row: EvalTestEntry) => number; }' to type 'EvalCollector' may be a mistake because neither type sufficiently overlaps with the other. If this was intentional, convert the expression to 'unknown' first. Type '{ addTest: (row: EvalTestEntry) => number; }' is missing the following properties from type 'EvalCollector': tier, tests, finalized, evalDir, and 6 more.": 3, "test/setup-gbrain-path4-caller.test.ts\tTS2339\tProperty 'diagnostic-secret' does not exist on type '{ 'no-verifier': number; 'no-install': number; 'no-init': number; 'no-registration': number; }'.": 1, "test/setup-gbrain-path4-caller.test.ts\tTS2339\tProperty 'exit' does not exist on type '{ 'no-verifier': number; 'no-install': number; 'no-init': number; 'no-registration': number; }'.": 1, "test/setup-gbrain-path4-caller.test.ts\tTS2339\tProperty 'leaked-claude-md' does not exist on type '{ 'no-verifier': number; 'no-install': number; 'no-init': number; 'no-registration': number; }'.": 1, diff --git a/test/helpers/eval-store.test.ts b/test/helpers/eval-store.test.ts index f613c3b42..8c55906c0 100644 --- a/test/helpers/eval-store.test.ts +++ b/test/helpers/eval-store.test.ts @@ -14,8 +14,20 @@ import { formatComparison, generateCommentary, judgePassed, + CONTRACT_VIOLATIONS_FILE, + ContractViolation, + TRIAL_ENV, + TRIAL_OUTCOME_SCHEMA, + expectContract, + failureClassOf, + formatTrialOutcomes, + panelVerdict, + parseTrialOutcomes, + sanitizeTrialError, + trialContextFromEnv, } from './eval-store'; -import type { EvalResult, EvalTestEntry, ComparisonResult } from './eval-store'; +import type { EvalResult, EvalTestEntry, ComparisonResult, PanelTrial, TrialOutcomeRecord } from './eval-store'; +import { EVAL_POLICY } from './periodic-exclude-data'; import { manualReviewFixture } from './manual-judge-review-fixture'; let tmpDir: string; @@ -957,3 +969,191 @@ describe('generateCommentary', () => { expect(notes.some(n => n.includes('Stable run'))).toBe(true); }); }); + +// --- Trials, panel verdicts and contract vetoes (eval reliability policy) --- + +const PANEL = EVAL_POLICY.panel; +const pass = (trial: number, extra: Partial = {}): PanelTrial => ({ trial, outcome: 'passed', ...extra }); +const fail = (trial: number, extra: Partial = {}): PanelTrial => ({ trial, outcome: 'failed', ...extra }); +const behavior = (trials: PanelTrial[], quarantined = false) => + panelVerdict({ case: 'case-x', kind: 'behavior', panel: PANEL, trials, quarantined }); + +describe('panelVerdict', () => { + test('policy constants are the approved pre-registration', () => { + expect(EVAL_POLICY.panel).toEqual({ n: 3, k: 2 }); + expect(EVAL_POLICY.quarantine).toEqual({ + entry: { rate: 0.95, minTrials: 10 }, + exit: { rate: 0.97, minTrials: 10 }, + capFraction: 0.1, + expiryWeeklyRuns: 8, + }); + expect(EVAL_POLICY.infraRedispatch).toBe(1); + }); + + test('rule: one trial, any failure fails the lane', () => { + const ok = panelVerdict({ case: 'r', kind: 'rule', panel: { n: 1, k: 1 }, trials: [pass(1)] }); + expect(ok).toMatchObject({ status: 'PASS', split: false, failsLane: false, coverage: true, marks: '✓' }); + const bad = panelVerdict({ case: 'r', kind: 'rule', panel: { n: 1, k: 1 }, trials: [fail(1)] }); + expect(bad).toMatchObject({ status: 'FAIL', failsLane: true, coverage: false, redClass: 'VERDICT', marks: '✗' }); + }); + + test('behavior 3/3 is a clean PASS', () => { + expect(behavior([pass(1), pass(2), pass(3)])).toMatchObject({ status: 'PASS', split: false, passed: 3, failsLane: false }); + }); + + test('behavior 2/3 is a split PASS that shows its failed trial', () => { + const v = behavior([pass(1), fail(2, { exit_reason: 'timeout' }), pass(3)]); + expect(v).toMatchObject({ status: 'PASS', split: true, passed: 2, failed: 1, failsLane: false, coverage: true, marks: '✓✗✓', reason: 'PASS 2/3' }); + expect(v.trials[1].exit_reason).toBe('timeout'); + }); + + test('behavior 1/3 and 0/3 fail the lane', () => { + expect(behavior([pass(1), fail(2), fail(3)])).toMatchObject({ status: 'FAIL', failsLane: true, redClass: 'VERDICT' }); + expect(behavior([fail(1), fail(2), fail(3)])).toMatchObject({ status: 'FAIL', failsLane: true }); + }); + + test('a contract trial fails the panel even at 2/3', () => { + const v = behavior([pass(1), pass(2), fail(3, { failure_class: 'contract' })]); + expect(v).toMatchObject({ status: 'FAIL', contract: true, failsLane: true, redClass: 'VERDICT', reason: 'contract violation' }); + }); + + test('a missing trial is INCOMPLETE and fails the lane', () => { + const v = behavior([pass(1), pass(3)]); + expect(v).toMatchObject({ status: 'INCOMPLETE', failsLane: true, coverage: false, redClass: 'INCOMPLETE', marks: '✓·✓' }); + expect(v.reason).toContain('missing trial t2'); + }); + + test('duplicate or out-of-range trial records are INCOMPLETE, never deduplicated', () => { + expect(behavior([pass(1), pass(2), pass(2), fail(3)]).status).toBe('INCOMPLETE'); + expect(behavior([pass(1), pass(2), pass(3), pass(4)]).reason).toContain('unexpected trial t4'); + const v = panelVerdict({ case: 'c', kind: 'behavior', panel: { n: 11, k: 6 }, trials: Array.from({ length: 10 }, (_, i) => pass(i + 2)) }); + expect(v.reason).toContain('missing trial t1'); + }); + + test('timeout and infra trials count as failed, never passing', () => { + const v = behavior([pass(1), fail(2, { exit_reason: 'timeout' }), fail(3, { failure_class: 'infra' })]); + expect(v).toMatchObject({ status: 'FAIL', passed: 1, failed: 2, failsLane: true, redClass: 'VERDICT' }); + const infra = behavior([pass(1), fail(2, { failure_class: 'infra' }), fail(3, { failure_class: 'infra' })]); + expect(infra).toMatchObject({ status: 'FAIL', redClass: 'INFRA' }); + expect(failureClassOf({ exit_reason: 'timeout' })).toBe('timeout'); + expect(failureClassOf({})).toBe('assertion'); + }); + + test('quarantined: 1/3 does not fail the lane, 0/3 and contract do, no coverage credit', () => { + expect(behavior([pass(1), fail(2), fail(3)], true)).toMatchObject({ status: 'FAIL', failsLane: false, coverage: false, redClass: null }); + expect(behavior([fail(1), fail(2), fail(3)], true)).toMatchObject({ status: 'FAIL', failsLane: true }); + expect(behavior([pass(1), pass(2), fail(3, { failure_class: 'contract' })], true)).toMatchObject({ status: 'FAIL', failsLane: true }); + expect(behavior([pass(1), pass(2), pass(3)], true)).toMatchObject({ status: 'PASS', coverage: false, failsLane: false }); + expect(behavior([pass(1), pass(2)], true)).toMatchObject({ status: 'INCOMPLETE', failsLane: true }); + }); + + test('quarantined rule keeps rule meaning (k = n)', () => { + const v = panelVerdict({ case: 'r', kind: 'rule', panel: { n: 3, k: 3 }, trials: [pass(1), pass(2), fail(3)], quarantined: true }); + expect(v).toMatchObject({ status: 'FAIL', failsLane: false }); + }); + + test('all-skipped panel is SKIPPED with no credit; partly skipped is INCOMPLETE', () => { + const skip = (trial: number): PanelTrial => ({ trial, outcome: 'skipped' }); + expect(behavior([skip(1), skip(2), skip(3)])).toMatchObject({ status: 'SKIPPED', coverage: false, failsLane: false }); + expect(behavior([pass(1), pass(2), skip(3)])).toMatchObject({ status: 'INCOMPLETE', failsLane: true }); + }); + + test('trials of different run attempts are never merged into one verdict', () => { + expect(() => behavior([pass(1), pass(2), fail(3, { attempt: 2 })])).toThrow(/run attempts/); + expect(behavior([pass(1, { attempt: 2 }), pass(2, { attempt: 2 }), pass(3, { attempt: 2 })]).attempt).toBe(2); + }); + + test('invalid panels throw', () => { + expect(() => panelVerdict({ case: 'c', kind: 'behavior', panel: { n: 3, k: 4 }, trials: [] })).toThrow(/invalid panel/); + expect(() => panelVerdict({ case: 'c', kind: 'nope' as any, panel: { n: 1, k: 1 }, trials: [] })).toThrow(/unknown kind/); + }); +}); + +describe('trial context and expectContract', () => { + let dir: string; + const saved: Record = {}; + const keys = [...Object.values(TRIAL_ENV), 'GSTACK_EVAL_DIR']; + beforeEach(() => { + dir = fs.mkdtempSync(path.join(os.tmpdir(), 'panel-verdict-')); + for (const key of keys) saved[key] = process.env[key]; + }); + afterEach(() => { + for (const key of keys) { + if (saved[key] === undefined) delete process.env[key]; + else process.env[key] = saved[key]; + } + fs.rmSync(dir, { recursive: true, force: true }); + }); + const setTrial = () => Object.assign(process.env, { + [TRIAL_ENV.caseId]: 'case-x', [TRIAL_ENV.kind]: 'behavior', [TRIAL_ENV.trial]: '2', + [TRIAL_ENV.panelN]: '3', [TRIAL_ENV.panelK]: '2', [TRIAL_ENV.policyVersion]: String(EVAL_POLICY.version), + GSTACK_EVAL_DIR: dir, + }); + + test('trialContextFromEnv: absent, complete, and malformed', () => { + for (const key of Object.values(TRIAL_ENV)) delete process.env[key]; + expect(trialContextFromEnv()).toBeNull(); + setTrial(); + expect(trialContextFromEnv()).toEqual({ case_id: 'case-x', kind: 'behavior', trial: 2, panel: { n: 3, k: 2 }, policy_version: EVAL_POLICY.version }); + process.env[TRIAL_ENV.trial] = '4'; + expect(() => trialContextFromEnv()).toThrow(/Malformed trial context/); + }); + + test('passing contract is a no-op', () => { + setTrial(); + expect(() => expectContract(true, 'fine')).not.toThrow(); + expect(fs.existsSync(path.join(dir, CONTRACT_VIOLATIONS_FILE))).toBe(false); + }); + + test('failed contract stamps the recorded entry and the sidecar before throwing', () => { + setTrial(); + const collector = new EvalCollector('e2e', dir); + collector.addTest({ name: 'case-x', suite: 's', tier: 'e2e', passed: true, duration_ms: 1, cost_usd: 0 }); + expect(() => expectContract(false, 'handoff missing', { collector, name: 'case-x' })).toThrow(ContractViolation); + const partial = JSON.parse(fs.readFileSync(path.join(dir, '_partial-e2e.json'), 'utf-8')); + expect(partial.tests[0]).toMatchObject({ passed: false, failure_class: 'contract', case_id: 'case-x', trial: 2, kind: 'behavior', panel: { n: 3, k: 2 } }); + const sidecar = fs.readFileSync(path.join(dir, CONTRACT_VIOLATIONS_FILE), 'utf-8').trim().split('\n').map((l) => JSON.parse(l)); + expect(sidecar).toEqual([expect.objectContaining({ case_id: 'case-x', trial: 2, message: 'handoff missing' })]); + }); + + test('a contract marked before recording stamps the later record, or becomes its own at finalize', async () => { + setTrial(); + const collector = new EvalCollector('e2e', dir); + expect(() => expectContract(0, 'no question asked', { collector, name: 'later' })).toThrow('CONTRACT: no question asked'); + collector.addTest({ name: 'later', suite: 's', tier: 'e2e', passed: true, duration_ms: 1, cost_usd: 0 }); + expect(() => expectContract(null, 'never recorded', { collector, name: 'orphan' })).toThrow(); + const file = await collector.finalize(); + const tests = JSON.parse(fs.readFileSync(file, 'utf-8')).tests; + expect(tests.find((t: any) => t.name === 'later')).toMatchObject({ passed: false, failure_class: 'contract' }); + expect(tests.find((t: any) => t.name === 'orphan')).toMatchObject({ passed: false, failure_class: 'contract', error: 'never recorded' }); + }); +}); + +describe('trial-outcomes JSONL', () => { + const record = (extra: Partial = {}): TrialOutcomeRecord => ({ + schema: TRIAL_OUTCOME_SCHEMA, case: 'case-x', file: 'test/x.test.ts', tier: 'gate', kind: 'behavior', + trial: 1, panel: { n: 3, k: 2 }, attempt: 1, outcome: 'passed', duration_ms: 10, cost_usd: 0.1, + policy_version: EVAL_POLICY.version, quarantined: false, execution: 'executed', source: 'shard', ...extra, + }); + + test('round-trips valid records', () => { + const records = [record(), record({ trial: 2, outcome: 'failed', failure_class: 'timeout', exit_reason: 'timeout' })]; + expect(parseTrialOutcomes(formatTrialOutcomes(records))).toEqual({ records, errors: [] }); + }); + + test('writer fails closed; reader reports bad lines as data errors', () => { + expect(() => formatTrialOutcomes([record({ outcome: 'failed' })])).toThrow(/failed without failure_class/); + expect(() => formatTrialOutcomes([record({ trial: 4 })])).toThrow(/trial invalid/); + const text = `${JSON.stringify(record())}\nnot json\n${JSON.stringify({ ...record(), schema: 'other' })}\n`; + const parsed = parseTrialOutcomes(text); + expect(parsed.records).toHaveLength(1); + expect(parsed.errors).toEqual(['line 2: not JSON', 'line 3: schema other']); + expect(parseTrialOutcomes(text, { maxBytes: 10 }).errors[0]).toContain('exceed'); + }); + + test('sanitizeTrialError keeps one capped line without mentions', () => { + expect(sanitizeTrialError('\n expected @garrytan to `see`\nsecond')).toBe("expected @\u200bgarrytan to 'see'"); + expect(sanitizeTrialError('x'.repeat(1000))!.length).toBe(300); + expect(sanitizeTrialError('')).toBeUndefined(); + }); +}); diff --git a/test/helpers/eval-store.ts b/test/helpers/eval-store.ts index b968447ea..32de29324 100644 --- a/test/helpers/eval-store.ts +++ b/test/helpers/eval-store.ts @@ -76,6 +76,18 @@ export interface EvalTestEntry { * its body again and re-records under the same name. Set by addTest. */ attempt?: number; + // Trial identity (eval reliability policy). Stamped by addTest from the + // TRIAL_ENV variables the paid runner sets on an isolated trial shard. + /** Registry id (E2E_TIERS / LLM_JUDGE_TOUCHFILES key) this record belongs to. */ + case_id?: string; + kind?: EvalCaseKind; + /** 1-based trial index within the case's panel. */ + trial?: number; + panel?: PanelShape; + /** Why a failed record failed; 'contract' comes only from expectContract. */ + failure_class?: TrialFailureClass; + policy_version?: number; + // E2E transcript?: any[]; prompt?: string; @@ -131,6 +143,326 @@ export function evalEntryOutcome(entry: unknown): 'passed' | 'failed' | 'manual- return result.passed === true ? 'passed' : 'failed'; } +// --- Trials and panel verdicts --- +// +// Paid evals never retry. Each case's kind (E2E_KINDS) fixes its trials before +// the run; a panel verdict is computed once, by panelVerdict(), from exactly +// panel.n trial records of one run attempt. The report, collector-outcomes, +// the PR comment and pass-rates all read that one function. + +export type EvalCaseKind = 'rule' | 'behavior' | 'judge'; +/** assertion: an ordinary failed expectation. contract: expectContract() fired + * (fails the panel at any count). timeout: the case budget ran out. + * infra: API/CLI/runner failure before the model could be graded. */ +export type TrialFailureClass = 'assertion' | 'contract' | 'timeout' | 'infra'; +export type TrialOutcome = 'passed' | 'failed' | 'skipped'; +export interface PanelShape { n: number; k: number } + +/** Environment the paid runner sets on an isolated trial shard. */ +export const TRIAL_ENV = { + caseId: 'GSTACK_EVAL_CASE_ID', + kind: 'GSTACK_EVAL_KIND', + trial: 'GSTACK_EVAL_TRIAL', + panelN: 'GSTACK_EVAL_PANEL_N', + panelK: 'GSTACK_EVAL_PANEL_K', + policyVersion: 'GSTACK_EVAL_POLICY_VERSION', +} as const; + +/** Sidecar every expectContract() failure appends to (in GSTACK_EVAL_DIR), so a + * contract veto survives a test that throws before recording its entry. */ +export const CONTRACT_VIOLATIONS_FILE = 'contract-violations.jsonl'; + +export interface TrialContext { + case_id: string; + kind: EvalCaseKind; + trial: number; + panel: PanelShape; + policy_version: number; +} + +const EVAL_KINDS: readonly EvalCaseKind[] = ['rule', 'behavior', 'judge']; +const FAILURE_CLASSES: readonly TrialFailureClass[] = ['assertion', 'contract', 'timeout', 'infra']; + +function positiveInt(raw: string | undefined): number | null { + if (raw === undefined || !/^[1-9][0-9]*$/.test(raw)) return null; + return Number(raw); +} + +/** Trial context of this process, or null outside an isolated trial shard. + * A partial or malformed context throws: a mislabeled trial is fail-open. */ +export function trialContextFromEnv(env: NodeJS.ProcessEnv = process.env): TrialContext | null { + const caseId = env[TRIAL_ENV.caseId]; + if (!caseId) return null; + const kind = env[TRIAL_ENV.kind] as EvalCaseKind | undefined; + const trial = positiveInt(env[TRIAL_ENV.trial]); + const n = positiveInt(env[TRIAL_ENV.panelN]); + const k = positiveInt(env[TRIAL_ENV.panelK]); + const policy = positiveInt(env[TRIAL_ENV.policyVersion]); + if (!kind || !EVAL_KINDS.includes(kind) || trial === null || n === null || k === null || policy === null || k > n || trial > n) { + throw new Error(`Malformed trial context for ${caseId}: ${Object.values(TRIAL_ENV).map((name) => `${name}=${env[name] ?? ''}`).join(' ')}`); + } + return { case_id: caseId, kind, trial, panel: { n, k }, policy_version: policy }; +} + +/** Failure class of a failed record: an explicit class wins, then the exit reason. */ +export function failureClassOf(entry: Pick): TrialFailureClass { + if (entry.failure_class && FAILURE_CLASSES.includes(entry.failure_class)) return entry.failure_class; + return entry.exit_reason === 'timeout' ? 'timeout' : 'assertion'; +} + +export class ContractViolation extends Error { + constructor(message: string) { + super(`CONTRACT: ${message}`); + this.name = 'ContractViolation'; + } +} + +/** + * Assert a contract: an outcome the product must meet on every run. On failure + * it records failure_class 'contract' before throwing, both on the collector + * entry named `record.name` (now or when the test records it) and in the + * GSTACK_EVAL_DIR sidecar, so panelVerdict() fails the panel even at 2 of 3. + */ +export function expectContract( + condition: unknown, + message: string, + record?: { collector: EvalCollector | null; name: string }, +): asserts condition { + if (condition) return; + record?.collector?.markContractViolation(record.name, message); + const evalDir = process.env.GSTACK_EVAL_DIR; + if (evalDir) { + const context = trialContextFromEnv(); + fs.mkdirSync(evalDir, { recursive: true }); + fs.appendFileSync(path.join(evalDir, CONTRACT_VIOLATIONS_FILE), JSON.stringify({ + case_id: context?.case_id ?? record?.name ?? null, + name: record?.name ?? null, + trial: context?.trial ?? null, + message, + at: new Date().toISOString(), + }) + '\n'); + } + throw new ContractViolation(message); +} + +export interface PanelTrial { + trial: number; + outcome: TrialOutcome; + /** Required meaning for a failed trial; absent reads as 'assertion'. */ + failure_class?: TrialFailureClass; + /** CI run attempt (github.run_attempt); absent means 1. */ + attempt?: number; + exit_reason?: string; + error?: string; + execution?: 'executed' | 'reused'; +} + +export interface PanelVerdictInput { + case: string; + kind: EvalCaseKind; + panel: PanelShape; + trials: readonly PanelTrial[]; + quarantined?: boolean; +} + +export type PanelStatus = 'PASS' | 'FAIL' | 'INCOMPLETE' | 'SKIPPED'; + +export interface PanelVerdict { + case: string; + kind: EvalCaseKind; + panel: PanelShape; + attempt: number; + quarantined: boolean; + status: PanelStatus; + passed: number; + failed: number; + /** A failed trial carried failure_class 'contract'. */ + contract: boolean; + /** PASS with at least one failed trial: shown as `PASS k/n`, never clean. */ + split: boolean; + /** Whether this verdict makes the lane red. */ + failsLane: boolean; + /** Whether it counts as passing coverage (never for quarantined or skipped). */ + coverage: boolean; + /** Machine classification of a lane-failing verdict: INCOMPLETE (missing or + * malformed trial records), INFRA (every failed trial is infra-class), or + * VERDICT (a real red). Null when the verdict does not fail the lane. */ + redClass: 'INCOMPLETE' | 'INFRA' | 'VERDICT' | null; + /** One glyph per trial index: ✓ pass, ✗ fail, – skipped, · missing. */ + marks: string; + reason: string; + trials: PanelTrial[]; +} + +/** + * The single verdict function. `rule`/`judge` cases run panel {1,1}; `behavior` + * cases run EVAL_POLICY.panel; a quarantined case runs a full panel whose k + * keeps its kind's meaning (k = n for rule). Verdict: INCOMPLETE unless + * exactly one record per trial index 1..n; SKIPPED when every trial skipped; + * FAIL on any contract trial; otherwise PASS iff passed >= k. A quarantined + * FAIL fails the lane only on a hard break (0 of n) or a contract violation. + */ +export function panelVerdict(input: PanelVerdictInput): PanelVerdict { + const { n, k } = input.panel; + if (!Number.isInteger(n) || !Number.isInteger(k) || n < 1 || k < 1 || k > n) { + throw new Error(`${input.case}: invalid panel {n:${n}, k:${k}}`); + } + if (!EVAL_KINDS.includes(input.kind)) throw new Error(`${input.case}: unknown kind ${String(input.kind)}`); + const attempts = new Set(input.trials.map((t) => t.attempt ?? 1)); + if (attempts.size > 1) { + throw new Error(`${input.case}: trials from run attempts ${[...attempts].join(', ')}; compute one verdict per attempt`); + } + const attempt = [...attempts][0] ?? 1; + const quarantined = input.quarantined === true; + const trials = [...input.trials].sort((a, b) => a.trial - b.trial); + const byIndex = new Map(); + const problems: string[] = []; + for (const t of trials) { + if (!Number.isInteger(t.trial) || t.trial < 1 || t.trial > n) problems.push(`unexpected trial t${t.trial}`); + else if (byIndex.has(t.trial)) problems.push(`duplicate trial t${t.trial}`); + else if (t.outcome !== 'passed' && t.outcome !== 'failed' && t.outcome !== 'skipped') problems.push(`t${t.trial} has outcome ${String(t.outcome)}`); + else byIndex.set(t.trial, t); + } + for (let i = 1; i <= n; i++) if (!trials.some((t) => t.trial === i)) problems.push(`missing trial t${i}`); + const marks = Array.from({ length: n }, (_, i) => { + const t = byIndex.get(i + 1); + return !t ? '·' : t.outcome === 'passed' ? '✓' : t.outcome === 'failed' ? '✗' : '–'; + }).join(''); + const passed = [...byIndex.values()].filter((t) => t.outcome === 'passed').length; + const failedTrials = [...byIndex.values()].filter((t) => t.outcome === 'failed'); + const skipped = [...byIndex.values()].filter((t) => t.outcome === 'skipped').length; + const contract = failedTrials.some((t) => failureClassOf(t) === 'contract'); + const base = { case: input.case, kind: input.kind, panel: { n, k }, attempt, quarantined, passed, failed: failedTrials.length, contract, marks, trials }; + + if (problems.length === 0 && skipped === n) { + return { ...base, status: 'SKIPPED', split: false, failsLane: false, coverage: false, redClass: null, reason: 'every trial skipped (no verdict credit)' }; + } + if (problems.length === 0 && skipped > 0) problems.push(`${skipped} of ${n} trials skipped`); + if (problems.length > 0) { + return { ...base, status: 'INCOMPLETE', split: false, failsLane: true, coverage: false, redClass: 'INCOMPLETE', reason: problems.join('; ') }; + } + if (!contract && passed >= k) { + const split = failedTrials.length > 0; + return { + ...base, status: 'PASS', split, failsLane: false, coverage: !quarantined, redClass: null, + reason: split ? `PASS ${passed}/${n}` : `${passed}/${n} passed`, + }; + } + const hardBreak = passed === 0; + const failsLane = !quarantined || contract || hardBreak; + const allInfra = !contract && failedTrials.length > 0 && failedTrials.every((t) => failureClassOf(t) === 'infra'); + const why = contract ? 'contract violation' : `${passed}/${n} passed, needs ${k}`; + return { + ...base, status: 'FAIL', split: false, failsLane, coverage: false, + redClass: failsLane ? (allInfra ? 'INFRA' : 'VERDICT') : null, + reason: !quarantined ? why + : contract ? `${why}; quarantine never excuses a contract` + : hardBreak ? `${why}; quarantined hard break` + : `${why}; quarantined, does not fail the lane`, + }; +} + +// --- trial-outcomes JSONL (one line per trial; pass-rate history input) --- + +export const TRIAL_OUTCOME_SCHEMA = 'gstack-trial-outcome/v1'; +export const TRIAL_OUTCOMES_FILE = 'trial-outcomes.jsonl'; +/** Cap on a stored `error` line (sanitized first line of the failure). */ +export const TRIAL_ERROR_MAX = 300; + +export interface TrialOutcomeRecord { + schema: typeof TRIAL_OUTCOME_SCHEMA; + /** Registry id. */ + case: string; + file: string; + tier: string; + kind: EvalCaseKind; + trial: number; + panel: PanelShape; + /** CI run attempt (github.run_attempt); 1 locally and for pre-policy backfill. */ + attempt: number; + outcome: TrialOutcome; + /** Present exactly when outcome is 'failed'. */ + failure_class?: TrialFailureClass; + exit_reason?: string; + error?: string; + duration_ms: number; + cost_usd: number; + model?: string; + cli_version?: string; + /** Reuse input key of the trial's shard, when known. */ + input_identity?: string; + /** EVAL_POLICY.version; 0 marks pre-policy backfill. */ + policy_version: number; + quarantined: boolean; + execution: 'executed' | 'reused'; + /** shard: isolated trial shard status. junit: a rule file shard's per-test + * JUnit outcome. backfill: imported pre-policy artifact record. */ + source: 'shard' | 'junit' | 'backfill'; + run_id?: string; + sha?: string; + lane?: string; + recorded_at?: string; +} + +/** First line of free text, stripped of @-mentions and control characters, capped. */ +export function sanitizeTrialError(text: string | undefined): string | undefined { + if (!text) return undefined; + const first = text.split('\n').map((l) => l.trim()).find((l) => l.length > 0); + if (!first) return undefined; + // eslint-disable-next-line no-control-regex + const clean = first.replace(/[\u0000-\u001f\u007f]/g, ' ').replace(/`/g, "'").replace(/@(?=[A-Za-z0-9_-])/g, '@\u200b'); + return clean.length > TRIAL_ERROR_MAX ? `${clean.slice(0, TRIAL_ERROR_MAX - 1)}…` : clean; +} + +function trialRecordProblems(r: any): string[] { + const problems: string[] = []; + if (!r || typeof r !== 'object' || Array.isArray(r)) return ['not an object']; + if (r.schema !== TRIAL_OUTCOME_SCHEMA) problems.push(`schema ${String(r.schema)}`); + for (const key of ['case', 'file', 'tier'] as const) if (typeof r[key] !== 'string' || r[key].length === 0) problems.push(`${key} missing`); + if (!EVAL_KINDS.includes(r.kind)) problems.push(`kind ${String(r.kind)}`); + const n = r.panel?.n, k = r.panel?.k; + if (!Number.isInteger(n) || !Number.isInteger(k) || n < 1 || k < 1 || k > n) problems.push('panel invalid'); + if (!Number.isInteger(r.trial) || r.trial < 1 || (Number.isInteger(n) && r.trial > n)) problems.push('trial invalid'); + if (!Number.isInteger(r.attempt) || r.attempt < 1) problems.push('attempt invalid'); + if (!['passed', 'failed', 'skipped'].includes(r.outcome)) problems.push(`outcome ${String(r.outcome)}`); + if (r.outcome === 'failed' && !FAILURE_CLASSES.includes(r.failure_class)) problems.push('failed without failure_class'); + if (r.outcome !== 'failed' && r.failure_class !== undefined) problems.push('failure_class on a non-failed trial'); + if (typeof r.duration_ms !== 'number' || !Number.isFinite(r.duration_ms) || r.duration_ms < 0) problems.push('duration_ms invalid'); + if (typeof r.cost_usd !== 'number' || !Number.isFinite(r.cost_usd) || r.cost_usd < 0) problems.push('cost_usd invalid'); + if (!Number.isInteger(r.policy_version) || r.policy_version < 0) problems.push('policy_version invalid'); + if (typeof r.quarantined !== 'boolean') problems.push('quarantined invalid'); + if (r.execution !== 'executed' && r.execution !== 'reused') problems.push('execution invalid'); + if (!['shard', 'junit', 'backfill'].includes(r.source)) problems.push('source invalid'); + if (r.error !== undefined && (typeof r.error !== 'string' || r.error.length > TRIAL_ERROR_MAX)) problems.push('error invalid'); + return problems; +} + +/** Serialize records as JSONL; throws on any invalid record (writers fail closed). */ +export function formatTrialOutcomes(records: readonly TrialOutcomeRecord[]): string { + return records.map((r) => { + const problems = trialRecordProblems(r); + if (problems.length > 0) throw new Error(`invalid trial record ${r?.case}~t${r?.trial}: ${problems.join(', ')}`); + return JSON.stringify(r); + }).join('\n') + (records.length > 0 ? '\n' : ''); +} + +/** Parse downloaded JSONL as data only: invalid lines are reported, never guessed. */ +export function parseTrialOutcomes(text: string, opts: { maxBytes?: number } = {}): { records: TrialOutcomeRecord[]; errors: string[] } { + const maxBytes = opts.maxBytes ?? 16 * 1024 * 1024; + if (Buffer.byteLength(text) > maxBytes) return { records: [], errors: [`trial outcomes exceed ${maxBytes} bytes`] }; + const records: TrialOutcomeRecord[] = []; + const errors: string[] = []; + text.split('\n').forEach((line, i) => { + if (line.trim() === '') return; + let parsed: unknown; + try { parsed = JSON.parse(line); } catch { errors.push(`line ${i + 1}: not JSON`); return; } + const problems = trialRecordProblems(parsed); + if (problems.length > 0) errors.push(`line ${i + 1}: ${problems.join(', ')}`); + else records.push(parsed as TrialOutcomeRecord); + }); + return { records, errors }; +} + export interface EvalResult { schema_version: number; version: string; @@ -887,6 +1219,7 @@ export class EvalCollector { private shard: string | null; private fileNamespace?: string; private createdAt = Date.now(); + private pendingContract = new Map(); constructor(tier: 'e2e' | 'llm-judge', evalDir?: string, fileNamespace?: string) { if (fileNamespace !== undefined && !/^[a-z0-9]+(?:-[a-z0-9]+)*$/.test(fileNamespace)) { @@ -903,7 +1236,29 @@ export class EvalCollector { // names are unique by convention). Stamp the 1-based attempt so a // pass-on-attempt-2 stays visible forever — the stream hides it. const prior = this.tests.filter((t) => t.name === entry.name).length; - this.tests.push({ ...entry, attempt: prior + 1 }); + const context = trialContextFromEnv(); + const record: EvalTestEntry = { ...(context ?? {}), ...entry, attempt: prior + 1 }; + const contract = this.pendingContract.get(entry.name); + if (contract !== undefined) { + this.pendingContract.delete(entry.name); + Object.assign(record, { passed: false, failure_class: 'contract', error: record.error ?? contract }); + } + this.tests.push(record); + this.savePartial(); + } + + /** expectContract() hook: mark `name`'s latest record (or its next one) as a + * contract failure. An unmatched mark becomes its own failed record at + * finalize, so the veto is never lost. */ + markContractViolation(name: string, message: string): void { + const existing = this.tests.filter((t) => t.name === name).at(-1); + if (!existing) { + this.pendingContract.set(name, message); + return; + } + existing.passed = false; + existing.failure_class = 'contract'; + existing.error = existing.error ?? message; this.savePartial(); } @@ -959,6 +1314,14 @@ export class EvalCollector { async finalize(): Promise { if (this.finalized) return ''; this.finalized = true; + for (const [name, message] of this.pendingContract) { + this.tests.push({ + ...(trialContextFromEnv() ?? {}), + name, suite: 'contract', tier: this.tier, passed: false, duration_ms: 0, cost_usd: 0, + failure_class: 'contract', error: message, attempt: 1, + }); + } + this.pendingContract.clear(); const git = getGitInfo(); const version = getVersion(); diff --git a/test/setup-gbrain-fixture.test.ts b/test/setup-gbrain-fixture.test.ts index 7053e095b..1abdb7ebd 100644 --- a/test/setup-gbrain-fixture.test.ts +++ b/test/setup-gbrain-fixture.test.ts @@ -169,7 +169,7 @@ describe('setup-gbrain owned Path 4 fixture', () => { pathToClaudeCodeExecutable: '/nonexistent/free-test-never-spawn-claude', signal: controller.signal, }, () => { validated = true; }, mode === 'deadline' ? 250 : 1000, { - collector: { addTest: (row: EvalTestEntry) => rows.push(row) } as EvalCollector, + collector: { addTest: (row: EvalTestEntry) => rows.push(row) } as unknown as EvalCollector, name: 'setup-gbrain-path4-local-pglite', suite: 'setup-gbrain', }); } catch (error) { failure = String(error); } @@ -214,7 +214,7 @@ describe('setup-gbrain owned Path 4 fixture', () => { await Promise.resolve(); controller.abort(new Error('caller cancelled during validation')); }, 1000, { - collector: { addTest: (row: EvalTestEntry) => rows.push(row) } as EvalCollector, + collector: { addTest: (row: EvalTestEntry) => rows.push(row) } as unknown as EvalCollector, name: 'setup-gbrain-path4-local-pglite', suite: 'setup-gbrain', })).rejects.toThrow('caller cancelled during validation'); expect(rows).toHaveLength(1); @@ -288,7 +288,7 @@ describe('setup-gbrain owned Path 4 fixture', () => { throw new Error(`assertion diagnostic ${fixture.token} ${credentialUrl}`); } }, undefined, { - collector: { addTest: (row: EvalTestEntry) => rows.push(row) } as EvalCollector, + collector: { addTest: (row: EvalTestEntry) => rows.push(row) } as unknown as EvalCollector, name: 'setup-gbrain-path4-local-pglite', suite: 'setup-gbrain', }); } catch (error) { thrown = String(error); } From a427f05c588d83843f357b7a12ea9422cc7e2ff7 Mon Sep 17 00:00:00 2001 From: garrytan Date: Tue, 29 Sep 2026 19:02:34 +0000 Subject: [PATCH 03/15] test(evals): pin the fail-closed rule-shard gate through the real --report path Synthetic slice artifacts for rule fail, timeout, missing slice, unreported entry, hollow, never-started, collector failure and wrong-slice reports all exit red before the panel-verdict gate change lands. --- test/paid-report-fail-open.test.ts | 110 +++++++++++++++++++++++++++++ 1 file changed, 110 insertions(+) create mode 100644 test/paid-report-fail-open.test.ts diff --git a/test/paid-report-fail-open.test.ts b/test/paid-report-fail-open.test.ts new file mode 100644 index 000000000..cc734f998 --- /dev/null +++ b/test/paid-report-fail-open.test.ts @@ -0,0 +1,110 @@ +/** + * Fail-open regression suite for the paid lane verdict. Synthetic slice + * artifacts go through the real `--report` CLI path (the command the workflow + * report jobs run), so a change to the gate cannot turn a real failure green + * without one of these cases going red. Landed before the panel-verdict gate + * change; every later gate change extends it. + */ +import { afterAll, beforeAll, describe, expect, test } from 'bun:test'; +import * as fs from 'node:fs'; +import * as os from 'node:os'; +import * as path from 'node:path'; +import { spawnSync } from 'node:child_process'; +import { parseRunManifest, type PaidRunManifest, type SliceResult } from '../scripts/test-paid-shards'; + +const ROOT = path.resolve(import.meta.dir, '..'); +const RULE_A = 'test/skill-e2e-fail-open-alpha.test.ts'; +const RULE_B = 'test/skill-e2e-fail-open-beta.test.ts'; + +let base: string; +beforeAll(() => { base = fs.mkdtempSync(path.join(os.tmpdir(), 'paid-fail-open-')); }); +afterAll(() => { fs.rmSync(base, { recursive: true, force: true }); }); + +type Outcome = SliceResult['outcomes'][number]; +const passed = (file: string, extra: Partial = {}): Outcome => + ({ files: [file], status: 'passed', exitCode: 0, elapsedMs: 1_000, executedTests: 1, skippedTests: 0, ...extra }); + +function manifest(entries: PaidRunManifest['entries'], sliceCount: number): PaidRunManifest { + return parseRunManifest(JSON.stringify({ version: 1, tier: 'periodic', evalsAll: true, sliceCount, + selectionReason: 'fail-open fixture', profile: 'full', selection: { e2e: null, judges: null }, entries })); +} + +let caseCounter = 0; +function report(plan: PaidRunManifest, slices: SliceResult[], collectors: Record = {}) { + const dir = path.join(base, `case-${++caseCounter}`); + fs.mkdirSync(dir, { recursive: true }); + fs.writeFileSync(path.join(dir, 'manifest.json'), JSON.stringify(plan)); + for (const slice of slices) fs.writeFileSync(path.join(dir, `slice-${slice.sliceIndex}.json`), JSON.stringify(slice)); + for (const [name, body] of Object.entries(collectors)) { + fs.mkdirSync(path.dirname(path.join(dir, name)), { recursive: true }); + fs.writeFileSync(path.join(dir, name), JSON.stringify(body)); + } + const result = spawnSync(process.execPath, [path.join(ROOT, 'scripts/test-paid-shards.ts'), '--tier', plan.tier, '--report', dir], + { cwd: ROOT, encoding: 'utf8', timeout: 30_000, env: { ...process.env, EVALS_TIER: plan.tier } }); + return { status: result.status, out: `${result.stdout}\n${result.stderr}`, dir }; +} + +const slice = (sliceIndex: number, sliceCount: number, outcomes: Outcome[]): SliceResult => + ({ version: 1, tier: 'periodic', profile: 'full', selection: { e2e: null, judges: null }, sliceIndex, sliceCount, outcomes }); + +describe('rule shards stay fail-closed through --report', () => { + const plan = manifest([ + { file: RULE_A, slice: 1, status: 'planned' }, + { file: RULE_B, slice: 2, status: 'planned' }, + ], 2); + + test('all planned rule shards passed: green', () => { + const r = report(plan, [slice(1, 2, [passed(RULE_A)]), slice(2, 2, [passed(RULE_B)])]); + expect(r.status, r.out).toBe(0); + }); + + test('a failed rule shard: red, naming the file', () => { + const r = report(plan, [slice(1, 2, [passed(RULE_A, { status: 'failed', exitCode: 1 })]), slice(2, 2, [passed(RULE_B)])]); + expect(r.status).toBe(1); + expect(r.out).toContain(`${RULE_A}: failed`); + }); + + test('a timed-out rule shard: red', () => { + const r = report(plan, [slice(1, 2, [passed(RULE_A, { status: 'timed-out', exitCode: null })]), slice(2, 2, [passed(RULE_B)])]); + expect(r.status).toBe(1); + expect(r.out).toContain(`${RULE_A}: timed-out`); + }); + + test('a missing slice artifact: red', () => { + const r = report(plan, [slice(1, 2, [passed(RULE_A)])]); + expect(r.status).toBe(1); + expect(r.out).toContain('slice 2/2 reported NO result'); + }); + + test('a planned shard no slice reported: red', () => { + const r = report(plan, [slice(1, 2, [passed(RULE_A)]), slice(2, 2, [])]); + expect(r.status).toBe(1); + expect(r.out).toContain(`planned ${RULE_B} (slice 2) was never reported`); + }); + + test('a hollow shard under EVALS_ALL: red', () => { + const r = report(plan, [slice(1, 2, [passed(RULE_A, { status: 'passed-empty', executedTests: 0 })]), slice(2, 2, [passed(RULE_B)])]); + expect(r.status).toBe(1); + expect(r.out).toContain(`${RULE_A}: passed-empty`); + }); + + test('a never-started shard: red', () => { + const r = report(plan, [slice(1, 2, [passed(RULE_A, { status: 'never-started', exitCode: null, executedTests: null })]), slice(2, 2, [passed(RULE_B)])]); + expect(r.status).toBe(1); + }); + + test('a failed collector record under a passing shard: red', () => { + const r = report(plan, [slice(1, 2, [passed(RULE_A)]), slice(2, 2, [passed(RULE_B)])], { + 'shards/skill-e2e-fail-open-alpha/run.json': { tier: 'e2e', total_tests: 1, total_cost_usd: 0, + tests: [{ name: 'alpha', suite: 's', tier: 'e2e', passed: false, duration_ms: 1, cost_usd: 0 }] }, + }); + expect(r.status).toBe(1); + expect(r.out).toContain('1 unapproved final collector failure(s)'); + }); + + test('a shard reported by the wrong slice: red', () => { + const r = report(plan, [slice(1, 2, [passed(RULE_A), passed(RULE_B)]), slice(2, 2, [])]); + expect(r.status).toBe(1); + expect(r.out).toContain(`${RULE_B} planned for slice 2 but reported by slice 1`); + }); +}); From b1f5bc0032cf694d5ae4167bd3a1eb99a6046350 Mon Sep 17 00:00:00 2001 From: garrytan Date: Tue, 29 Sep 2026 19:09:51 +0000 Subject: [PATCH 04/15] test(evals): retire every paid automatic retry Paid evals never retry (approved 2026-09-29): delete SHORT_CASE_RETRY_FILES and retriesWithinCaseCap, drop the retry fields from the registered wall rows (walls now cover one run plus reserve), make retriesForFiles return 0, pass --retry 0 explicitly, and drop --retry 1 from the package.json paid scripts. Add the eval:pass-rates alias. Tests that pinned the old retry allowance are updated as a policy change; review-finalization-budget now proves late-result recording under the production zero-retry arguments. --- package.json | 13 +- scripts/test-paid-shards.ts | 31 ++--- test/carve-section-sharding.test.ts | 2 +- test/eng-finding-retry-budget.test.ts | 13 +- test/helpers/eval-budgets.ts | 148 ++++++++--------------- test/paid-overlay-scheduling.test.ts | 12 +- test/paid-retry-supervision.test.ts | 76 +++++------- test/paid-run-manifest.test.ts | 13 +- test/review-finalization-budget.test.ts | 20 ++- test/ship-hook-actor.test.ts | 4 +- test/ship-skip-actor.test.ts | 4 +- test/workflow-boundaries-fixture.test.ts | 4 +- 12 files changed, 131 insertions(+), 209 deletions(-) diff --git a/package.json b/package.json index fb8002dbd..86ddbcbc1 100644 --- a/package.json +++ b/package.json @@ -31,12 +31,12 @@ "test:free": "bun run scripts/test-free-shards.ts", "test:windows": "bun run scripts/test-free-shards.ts --windows-only", "test:ubicloud": "bash scripts/ubicloud/test-free.sh", - "test:evals": "EVALS=1 bun test --retry 1 --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-llm-eval*.test.ts test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/llm-judge-recommendation.test.ts test/carve-section-loading*.test.ts", - "test:evals:all": "EVALS=1 EVALS_ALL=1 bun test --retry 1 --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-llm-eval*.test.ts test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/llm-judge-recommendation.test.ts test/carve-section-loading*.test.ts", - "test:e2e": "EVALS=1 bun test --retry 1 --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/carve-section-loading*.test.ts", - "test:e2e:all": "EVALS=1 EVALS_ALL=1 bun test --retry 1 --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/carve-section-loading*.test.ts", - "test:gate": "EVALS=1 EVALS_TIER=gate bun test --retry 1 --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-llm-eval*.test.ts test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/llm-judge-recommendation.test.ts test/carve-section-loading*.test.ts", - "test:periodic": "EVALS=1 EVALS_TIER=periodic EVALS_ALL=1 bun test --retry 1 --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-llm-eval*.test.ts test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/llm-judge-recommendation.test.ts test/carve-section-loading*.test.ts", + "test:evals": "EVALS=1 bun test --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-llm-eval*.test.ts test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/llm-judge-recommendation.test.ts test/carve-section-loading*.test.ts", + "test:evals:all": "EVALS=1 EVALS_ALL=1 bun test --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-llm-eval*.test.ts test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/llm-judge-recommendation.test.ts test/carve-section-loading*.test.ts", + "test:e2e": "EVALS=1 bun test --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/carve-section-loading*.test.ts", + "test:e2e:all": "EVALS=1 EVALS_ALL=1 bun test --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/carve-section-loading*.test.ts", + "test:gate": "EVALS=1 EVALS_TIER=gate bun test --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-llm-eval*.test.ts test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/llm-judge-recommendation.test.ts test/carve-section-loading*.test.ts", + "test:periodic": "EVALS=1 EVALS_TIER=periodic EVALS_ALL=1 bun test --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-llm-eval*.test.ts test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/llm-judge-recommendation.test.ts test/carve-section-loading*.test.ts", "test:gate:sharded": "bun run scripts/test-paid-shards.ts --tier gate", "test:periodic:sharded": "EVALS_ALL=1 bun run scripts/test-paid-shards.ts --tier periodic", "test:codex": "EVALS=1 bun test test/codex-e2e.test.ts test/codex-e2e-sol-scope.test.ts", @@ -52,6 +52,7 @@ "eval:compare": "bun run scripts/eval-compare.ts", "eval:summary": "bun run scripts/eval-summary.ts", "eval:flake-rank": "bun run scripts/eval-flake-rank.ts", + "eval:pass-rates": "bun run scripts/eval-flake-rank.ts", "eval:watch": "bun run scripts/eval-watch.ts", "eval:select": "bun run scripts/eval-select.ts", "analytics": "bun run scripts/analytics.ts", diff --git a/scripts/test-paid-shards.ts b/scripts/test-paid-shards.ts index 8790ac71d..c4f7031b8 100644 --- a/scripts/test-paid-shards.ts +++ b/scripts/test-paid-shards.ts @@ -67,7 +67,7 @@ import { } from './test-strict-output'; import { PAID_TEST_GLOBS, isPaidTestFile } from '../test/helpers/paid-test-set'; import { CASE_CI_EXCLUDE, PERIODIC_CI_EXCLUDE } from '../test/helpers/periodic-exclude-data'; -import { FILE_RETRY_BUDGETS, SHORT_CASE_RETRY_FILES, STRICT_RETRY_CASE_BUDGETS } from '../test/helpers/eval-budgets'; +import { FILE_RETRY_BUDGETS, STRICT_RETRY_CASE_BUDGETS } from '../test/helpers/eval-budgets'; import { getProjectEvalDir, getClaudeCliVersion, isFinalizedEvalResultFile, evalEntryOutcome } from '../test/helpers/eval-store'; import { manualReviewProblem } from '../test/helpers/cookie-workflow-manual-review'; import { preflightAnthropicApi } from '../test/helpers/anthropic-preflight'; @@ -644,9 +644,9 @@ export function resolvePaidShardBudget(files: string[], overrideMs?: number): Pa if (overlay && overrideMs !== undefined && overrideMs < OVERLAY_MIN_FILE_WALL_MS) { throw new Error(`Overlay shard requires at least ${OVERLAY_MIN_FILE_WALL_MS}ms; explicit wall ${overrideMs}ms cannot preserve its work and finalization budget`); } - // A registered file's case shard supervises one case and its allowed attempts. + // A registered file's case shard supervises its one case. const registeredMs = finding && shardCaseId(files[0]!) !== null - ? finding.caseMs * (finding.retries + 1) + finding.shardReserveMs : finding?.shardMs; + ? finding.caseMs + finding.shardReserveMs : finding?.shardMs; return { timeoutMs: overrideMs ?? (registeredMs ?? (overlay ? OVERLAY_MIN_FILE_WALL_MS : DEFAULT_SHARD_TIMEOUT_MS)), source: overrideMs !== undefined ? 'explicit' : finding ? 'registered' : 'default', @@ -667,9 +667,9 @@ export function buildPaidShardArgs( // Explicit --concurrent/--max-concurrency: the legacy path always set one; // omitting it here made within-shard parallelism differ silently between // the two runners (observed: 1.6x sumdur/wall sharded vs 8x legacy). - // Retries come from retriesForFiles (the timeout-is-a-verdict rule) at the - // call site; the fallback of 1 serves only direct callers. - return ['test', ...files, '--retry', String(retries ?? 1), '--concurrent', `--max-concurrency=${maxConcurrency}`, `--timeout=${timeoutMs}`]; + // Paid evals never retry (retriesForFiles); `--retry 0` is explicit so a + // bunfig default can never reintroduce one. + return ['test', ...files, '--retry', String(retries ?? 0), '--concurrent', `--max-concurrency=${maxConcurrency}`, `--timeout=${timeoutMs}`]; } /** @@ -1254,20 +1254,13 @@ export interface PaidRunManifest { } /** - * Automatic retries follow the approved rule in test/helpers/eval-budgets.ts: - * a timed-out attempt is a verdict, so only files whose every case budget is - * at most RETRY_MAX_CASE_MS keep a retry (registered rows derive it from their - * caseMs; SHORT_CASE_RETRY_FILES lists the rest). Overlays and every other - * paid file run once. A multi-file shard takes the smallest allowance. + * Paid evals never retry (approved 2026-09-29): a failed verdict is final for + * its run, and trials are fixed by kind before the run (EVAL_POLICY). The + * function stays the single statement of that policy for the Bun arguments + * and the reuse identity. */ -export function retriesForFiles(files: string[]): number { - if (files.some(isOverlayTestFile)) return 0; - return Math.min(...files.map((file) => { - const rel = shardFile(file); - const registered = FILE_RETRY_BUDGETS.find(budget => budget.file === rel); - if (registered) return registered.retries; - return SHORT_CASE_RETRY_FILES.includes(rel) ? 1 : 0; - })); +export function retriesForFiles(_files: string[]): number { + return 0; } export const PAID_TEST_DURATIONS_FILE = 'scripts/paid-test-durations.json'; diff --git a/test/carve-section-sharding.test.ts b/test/carve-section-sharding.test.ts index 4f60cfa43..e5710c166 100644 --- a/test/carve-section-sharding.test.ts +++ b/test/carve-section-sharding.test.ts @@ -22,7 +22,7 @@ describe('carved-skill cases each get a complete paid process budget', () => { expect(selectPaidTestFiles(files.map(file => 'test/' + file), 'periodic').selected).toHaveLength(files.length); expect(selectPaidTestFiles(files.map(file => 'test/' + file), 'gate').selected).toHaveLength(0); }); - test('all configured retries plus teardown fit even with within-shard concurrency one', () => { + test('every case run plus teardown fits even with within-shard concurrency one', () => { for (const file of files) { const attempts = retriesForFiles(['test/' + file]) + 1; expect(CAPTURE_LONG_MS * attempts + 10_000).toBeLessThan(DEFAULT_SHARD_TIMEOUT_MS); diff --git a/test/eng-finding-retry-budget.test.ts b/test/eng-finding-retry-budget.test.ts index 4b313e03b..a865fd8a3 100644 --- a/test/eng-finding-retry-budget.test.ts +++ b/test/eng-finding-retry-budget.test.ts @@ -6,13 +6,12 @@ import os from 'node:os'; import path from 'node:path'; for (const budget of FINDING_RETRY_BUDGETS) { - test(`${budget.file}: supervision preserves every existing attempt and retry`, () => { + test(`${budget.file}: supervision covers its one run of every case`, () => { expect(budget.testMs).toBe(1_500_000); - // A 25-minute case is past RETRY_MAX_CASE_MS: a timed-out attempt is its verdict. - expect(budget.retries).toBe(0); - expect(retriesForFiles([budget.file])).toBe(budget.retries); + // Paid evals never retry: a timed-out case is its verdict. + expect(retriesForFiles([budget.file])).toBe(0); expect(budget.shardReserveMs).toBe(SHARD_RESERVE_MS); - expect(budget.shardMs).toBe(budget.cases * budget.testMs * (budget.retries + 1) + budget.shardReserveMs); + expect(budget.shardMs).toBe(budget.cases * budget.testMs + budget.shardReserveMs); expect(resolvePaidShardBudget([budget.file])).toEqual({ timeoutMs: budget.shardMs, source: 'registered', policyId: budget.id }); const source = fs.readFileSync(path.join(import.meta.dir, '..', budget.file), 'utf8'); if (budget.file === 'test/skill-e2e-plan-ceo-split-overflow.test.ts') { @@ -184,10 +183,10 @@ test('current detach supervision covers the live-census floor', () => { const pkg = JSON.parse(fs.readFileSync(path.join(import.meta.dir, '../package.json'), 'utf8')); const periodicTimeout = Number(pkg.scripts['eval:bg:periodic'].match(/--timeout\s+(\d+)/)[1]); const gateTimeout = Number(pkg.scripts['eval:bg:gate'].match(/--timeout\s+(\d+)/)[1]); - expect(floorFor('gate')).toBe(26_471); + expect(floorFor('gate')).toBe(21_725); expect(gateTimeout).toBe(49_320); expect(gateTimeout).toBeGreaterThanOrEqual(floorFor('gate')); - expect(floorFor('periodic')).toBe(30_797); + expect(floorFor('periodic')).toBe(22_481); }); for (const jobs of [1, 2, 3]) test(`FIFO bound covers partial durations with ${jobs} workers`, () => { diff --git a/test/helpers/eval-budgets.ts b/test/helpers/eval-budgets.ts index 023e6e4ac..a195c6f99 100644 --- a/test/helpers/eval-budgets.ts +++ b/test/helpers/eval-budgets.ts @@ -47,132 +47,80 @@ export const ALL_TIERS = { export const SHARD_RESERVE_MS = 2 * 60_000; /** - * Retry policy (approved 2026-09-29): a timed-out attempt is a verdict. Bun's - * --retry reruns a failed case after it may have spent its whole budget, so an - * automatic retry is kept only where one more attempt is short: every case of - * the file has a per-attempt budget of at most RETRY_MAX_CASE_MS, the CAPTURE - * tier plus its recording grace. Those failures are fast flake classes (API - * blips, tool hiccups) and a retry costs at most one more short attempt. Files - * with any longer case run once. Per-case budgets never change with this rule. + * Retry policy (approved 2026-09-29, eval reliability wave): paid evals never + * retry. Each case's kind (E2E_KINDS) fixes its trials before the run: `rule` + * one trial, `behavior` a panel of EVAL_POLICY.panel independent trials, and + * `judge` one case that samples its judge panel internally. A failed verdict + * is final for that run; a manual re-run adds trials under a new run attempt + * and never replaces the original verdict. Rows below keep only wall + * supervision; per-case budgets never change with this rule. */ -export const RETRY_MAX_CASE_MS = CAPTURE_MS + 15_000; -export function retriesWithinCaseCap(caseMs: number, configuredRetries: number): number { - return caseMs <= RETRY_MAX_CASE_MS ? configuredRetries : 0; -} - -/** - * Unregistered paid files that keep one automatic retry: every case budget is - * JUDGE or CAPTURE tier (test/paid-retry-supervision.test.ts scans each source). - * Registered rows below derive retries from their declared caseMs; every other - * paid file runs once. - */ -export const SHORT_CASE_RETRY_FILES: readonly string[] = [ - 'test/codex-e2e-sol-scope.test.ts', - 'test/llm-judge-recommendation.test.ts', - 'test/skill-e2e-ask-user-question-format-compliance.test.ts', - 'test/skill-e2e-benchmark-providers.test.ts', - 'test/skill-e2e-bws.test.ts', - 'test/skill-e2e-context-skills.test.ts', - 'test/skill-e2e-coverage-audit.test.ts', - 'test/skill-e2e-diagram.test.ts', - 'test/skill-e2e-first-task-scaffold.test.ts', - 'test/skill-e2e-gbrain-roundtrip-local.test.ts', - 'test/skill-e2e-hermetic-canary.test.ts', - 'test/skill-e2e-investigate-owned-completion.test.ts', - 'test/skill-e2e-investigate-owned-termination.test.ts', - 'test/skill-e2e-learnings.test.ts', - 'test/skill-e2e-plan-tune.test.ts', - 'test/skill-e2e-qa-functional-fix.test.ts', - 'test/skill-e2e-qa-functional.test.ts', - 'test/skill-e2e-review-army.test.ts', - 'test/skill-e2e-review.test.ts', - 'test/skill-e2e-session-intelligence.test.ts', - 'test/skill-e2e-setup-gbrain-bad-token.test.ts', - 'test/skill-e2e-setup-gbrain-path4-local-pglite.test.ts', - 'test/skill-e2e-setup-gbrain-remote.test.ts', - 'test/skill-e2e-ship-hook-consent.test.ts', - 'test/skill-e2e-ship-hook-refresh.test.ts', - 'test/skill-e2e-ship-skip.test.ts', - 'test/skill-e2e-sync-gbrain-readiness.test.ts', - 'test/skill-e2e-third-party-actions.test.ts', - 'test/skill-e2e-triage.test.ts', - 'test/skill-routing-e2e.test.ts', -]; - -/** Whole-file supervision covers every attempt the retry policy allows. - * These fixtures allow 25 minutes per case, so they run once. +/** Whole-file supervision for one run of every case. + * These fixtures allow 25 minutes per case. * Reserve the sequential upper bound even when Bun runs sibling cases together. */ export const FINDING_RETRY_BUDGETS = [ { file: 'test/skill-e2e-plan-ceo-split-overflow.test.ts', cases: 1 }, { file: 'test/skill-e2e-plan-eng-multi-finding-batching.test.ts', cases: 1 }, -].map(({ file, cases }) => { - const retries = retriesWithinCaseCap(1_500_000, 1); - return { - file, cases, - id: `${file.slice('test/skill-e2e-'.length, -'.test.ts'.length)}-existing-retry-v1`, - testMs: 1_500_000, - caseMs: 1_500_000, - retries, - shardReserveMs: SHARD_RESERVE_MS, - shardMs: cases * 1_500_000 * (retries + 1) + SHARD_RESERVE_MS, - }; -}); +].map(({ file, cases }) => ({ + file, cases, + id: `${file.slice('test/skill-e2e-'.length, -'.test.ts'.length)}-existing-retry-v1`, + testMs: 1_500_000, + caseMs: 1_500_000, + shardReserveMs: SHARD_RESERVE_MS, + shardMs: cases * 1_500_000 + SHARD_RESERVE_MS, +})); -/** Three existing captures in one 16-minute case, so the file runs once. */ +/** Three existing captures in one 16-minute case. */ export const AUQ_CONSISTENCY_RETRY_BUDGET = { file: 'test/skill-e2e-auq-consistency.test.ts', id: 'auq-consistency-existing-retry-v1', cases: 1, testMs: 3 * CAPTURE_MS + 60_000, caseMs: 3 * CAPTURE_MS + 60_000, - retries: retriesWithinCaseCap(3 * CAPTURE_MS + 60_000, 1), shardReserveMs: SHARD_RESERVE_MS, - shardMs: (3 * CAPTURE_MS + 60_000) * (retriesWithinCaseCap(3 * CAPTURE_MS + 60_000, 1) + 1) + SHARD_RESERVE_MS, + shardMs: 3 * CAPTURE_MS + 60_000 + SHARD_RESERVE_MS, } as const; /** These fixtures have a fixed case count in every supported tier. */ export const STRICT_RETRY_CASE_BUDGETS = [...FINDING_RETRY_BUDGETS, AUQ_CONSISTENCY_RETRY_BUDGET]; -/** Whole-file walls cover all existing cases and every allowed attempt, even if - * Bun runs them sequentially. Mixed-tier files reserve their larger complete - * tier, never a currently selected subset. caseMs is the longest single case - * budget, which decides the retry (RETRY_MAX_CASE_MS). These rows add no - * case-count or model-work policy. The 10-second terms preserve the existing - * Codex/recording finalization grace. +/** Whole-file walls cover all existing cases, even if Bun runs them + * sequentially. Mixed-tier files reserve their larger complete tier, never a + * currently selected subset. caseMs is the longest single case budget, the + * wall of one isolated case shard. These rows add no case-count or model-work + * policy. The 10-second terms preserve the existing Codex/recording + * finalization grace. */ export const FILE_RETRY_BUDGETS = [ ...STRICT_RETRY_CASE_BUDGETS, ...[ - { file: 'test/skill-e2e-qa-callers.test.ts', attemptMs: 5 * (CAPTURE_MS + 15_000), caseMs: CAPTURE_MS + 15_000, configuredRetries: 1 }, - { file: 'test/skill-e2e-shared-libs-paths.test.ts', attemptMs: 3 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 1 }, - { file: 'test/skill-e2e-ship-docsync.test.ts', attemptMs: 4 * CAPTURE_LONG_MS + 8 * CAPTURE_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 1 }, + { file: 'test/skill-e2e-qa-callers.test.ts', attemptMs: 5 * (CAPTURE_MS + 15_000), caseMs: CAPTURE_MS + 15_000 }, + { file: 'test/skill-e2e-shared-libs-paths.test.ts', attemptMs: 3 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS }, + { file: 'test/skill-e2e-ship-docsync.test.ts', attemptMs: 4 * CAPTURE_LONG_MS + 8 * CAPTURE_MS, caseMs: CAPTURE_LONG_MS }, // Seventeen workflow judges include their 10s recording grace; the other - // seven judges retain 120s. Supervise all 24 and the existing one retry. - { file: 'test/skill-llm-eval.test.ts', attemptMs: 17 * (JUDGE_MS + 10_000) + 7 * JUDGE_MS, caseMs: JUDGE_MS + 10_000, configuredRetries: 1 }, - { file: 'test/skill-e2e-auq-matrix.test.ts', attemptMs: 6 * CAPTURE_MS, caseMs: CAPTURE_MS, configuredRetries: 1 }, - { file: 'test/skill-e2e-plan-format.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), caseMs: CAPTURE_MS + 10_000, configuredRetries: 1 }, - { file: 'test/skill-e2e-auto-decide-preserved.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 }, - { file: 'test/skill-e2e-plan-ceo-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 }, - { file: 'test/skill-e2e-plan-eng-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 }, - { file: 'test/skill-e2e-plan-design-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 }, - { file: 'test/skill-e2e-plan-devex-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 }, - { file: 'test/skill-e2e-plan-mode-no-op.test.ts', attemptMs: 5 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 2 }, - { file: 'test/skill-e2e-plan-ceo-mode-routing.test.ts', attemptMs: 2 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 1 }, - { file: 'test/skill-e2e-plan-eng-plan-mode.test.ts', attemptMs: 2 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 1 }, - { file: 'test/skill-e2e-plan-prosons.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), caseMs: CAPTURE_MS + 10_000, configuredRetries: 1 }, + // seven judges retain 120s. Supervise all 24. + { file: 'test/skill-llm-eval.test.ts', attemptMs: 17 * (JUDGE_MS + 10_000) + 7 * JUDGE_MS, caseMs: JUDGE_MS + 10_000 }, + { file: 'test/skill-e2e-auq-matrix.test.ts', attemptMs: 6 * CAPTURE_MS, caseMs: CAPTURE_MS }, + { file: 'test/skill-e2e-plan-format.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), caseMs: CAPTURE_MS + 10_000 }, + { file: 'test/skill-e2e-auto-decide-preserved.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS }, + { file: 'test/skill-e2e-plan-ceo-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS }, + { file: 'test/skill-e2e-plan-eng-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS }, + { file: 'test/skill-e2e-plan-design-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS }, + { file: 'test/skill-e2e-plan-devex-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS }, + { file: 'test/skill-e2e-plan-mode-no-op.test.ts', attemptMs: 5 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS }, + { file: 'test/skill-e2e-plan-ceo-mode-routing.test.ts', attemptMs: 2 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS }, + { file: 'test/skill-e2e-plan-eng-plan-mode.test.ts', attemptMs: 2 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS }, + { file: 'test/skill-e2e-plan-prosons.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), caseMs: CAPTURE_MS + 10_000 }, // Gate: six 300s cases + one 610s case; periodic: two 900s + three 600s. - { file: 'test/skill-e2e-plan.test.ts', attemptMs: Math.max(6 * CAPTURE_MS + CAPTURE_LONG_MS + 10_000, 2 * PTY_MS + 3 * CAPTURE_LONG_MS), caseMs: PTY_MS, configuredRetries: 1 }, - ].map(({ file, attemptMs, caseMs, configuredRetries }) => { - const retries = retriesWithinCaseCap(caseMs, configuredRetries); - return { - file, attemptMs, caseMs, retries, - id: `${file.slice('test/'.length, -'.test.ts'.length)}-existing-retry-v1`, - shardReserveMs: SHARD_RESERVE_MS, - shardMs: attemptMs * (retries + 1) + SHARD_RESERVE_MS, - }; - }), + { file: 'test/skill-e2e-plan.test.ts', attemptMs: Math.max(6 * CAPTURE_MS + CAPTURE_LONG_MS + 10_000, 2 * PTY_MS + 3 * CAPTURE_LONG_MS), caseMs: PTY_MS }, + ].map(({ file, attemptMs, caseMs }) => ({ + file, attemptMs, caseMs, + id: `${file.slice('test/'.length, -'.test.ts'.length)}-existing-retry-v1`, + shardReserveMs: SHARD_RESERVE_MS, + shardMs: attemptMs + SHARD_RESERVE_MS, + })), ]; /** No paid test may exceed the ordinary tiers; arbitrary per-file escapes fail. */ diff --git a/test/paid-overlay-scheduling.test.ts b/test/paid-overlay-scheduling.test.ts index 701eb5c07..7b08b466c 100644 --- a/test/paid-overlay-scheduling.test.ts +++ b/test/paid-overlay-scheduling.test.ts @@ -21,8 +21,8 @@ const fakeEnv = { }; describe('overlay file policy', () => { - test('grouped planning isolates every overlay and preserves ordinary retries', () => { - // Two short-case files keep their one retry (timeout-is-a-verdict rule). + test('grouped planning isolates every overlay and never retries ordinary files', () => { + // Paid evals never retry (approved 2026-09-29), short-case files included. const workflow = 'test/skill-e2e-review.test.ts'; const files = [...overlayFiles, 'test/skill-e2e-triage.test.ts', workflow]; for (const maxFilesPerShard of [2, 3, 10]) { @@ -31,9 +31,9 @@ describe('overlay file policy', () => { for (const file of overlayFiles) expect(shards).toContainEqual([file]); const workflowShard = shards.find(shard => shard.includes(workflow))!; expect(workflowShard.some(isOverlayTestFile)).toBe(false); - expect(retriesForFiles(workflowShard)).toBe(1); + expect(retriesForFiles(workflowShard)).toBe(0); const args = buildPaidShardArgs(workflowShard, resolvePaidShardTimeoutMs(workflowShard), 2, retriesForFiles(workflowShard)); - expect(args[args.indexOf('--retry') + 1]).toBe('1'); + expect(args[args.indexOf('--retry') + 1]).toBe('0'); expect(planPaidShards(files.map(file => file.replaceAll('/', '\\')), { maxFilesPerShard })).toEqual(shards); } }); @@ -66,10 +66,10 @@ describe('overlay file policy', () => { for (const file of [normalFile, 'test/skill-e2e-overlay-harness.test.ts', 'test/model-overlays.test.ts']) { expect(isOverlayTestFile(file)).toBe(false); expect(resolvePaidShardTimeoutMs([file])).toBe(DEFAULT_SHARD_TIMEOUT_MS); - // Not overlays; unlisted files run once because their case budget is unknown. + // Not overlays; every paid file runs once. expect(retriesForFiles([file])).toBe(0); } - expect(retriesForFiles(['test/skill-e2e-review.test.ts'])).toBe(1); + expect(retriesForFiles(['test/skill-e2e-review.test.ts'])).toBe(0); expect(resolvePaidShardTimeoutMs([normalFile], 1234)).toBe(1234); expect(resolvePaidShardTimeoutMs([overlayFiles[0]], 1_900_000)).toBe(1_900_000); expect(() => resolvePaidShardTimeoutMs([overlayFiles[0]], 1_800_000)).toThrow('explicit wall'); diff --git a/test/paid-retry-supervision.test.ts b/test/paid-retry-supervision.test.ts index 4cf8eb92a..cb989a9c2 100644 --- a/test/paid-retry-supervision.test.ts +++ b/test/paid-retry-supervision.test.ts @@ -7,7 +7,7 @@ import { shardFile, sliceExecutionOrder, sliceSupervisedWallMs, CASE_SHARDED_FILES, } from '../scripts/test-paid-shards'; import { - ALL_TIERS, AUQ_CONSISTENCY_RETRY_BUDGET, FILE_RETRY_BUDGETS, RETRY_MAX_CASE_MS, SHORT_CASE_RETRY_FILES, + ALL_TIERS, AUQ_CONSISTENCY_RETRY_BUDGET, FILE_RETRY_BUDGETS, FINDING_RETRY_BUDGETS, STRICT_RETRY_CASE_BUDGETS, } from './helpers/eval-budgets'; @@ -15,16 +15,16 @@ import { E2E_TOUCHFILES } from './helpers/touchfiles'; const read = (file: string) => readFileSync(join(import.meta.dir, '..', file), 'utf8'); const newBudgets = FILE_RETRY_BUDGETS.filter(row => !FINDING_RETRY_BUDGETS.some(old => old.file === row.file)); -// Walls cover every attempt the retry rule allows: files with a case budget -// past RETRY_MAX_CASE_MS run once (a timed-out attempt is a verdict). +// Paid evals never retry (approved 2026-09-29): each wall covers one run of +// every case plus the supervision reserve. const expectedWalls = { - 'test/skill-e2e-qa-callers.test.ts': 3_270_000, + 'test/skill-e2e-qa-callers.test.ts': 1_695_000, 'test/skill-e2e-shared-libs-paths.test.ts': 1_920_000, 'test/skill-e2e-ship-docsync.test.ts': 4_920_000, - 'test/skill-llm-eval.test.ts': 6_220_000, + 'test/skill-llm-eval.test.ts': 3_170_000, 'test/skill-e2e-auq-consistency.test.ts': 1_080_000, - 'test/skill-e2e-auq-matrix.test.ts': 3_720_000, - 'test/skill-e2e-plan-format.test.ts': 2_600_000, + 'test/skill-e2e-auq-matrix.test.ts': 1_920_000, + 'test/skill-e2e-plan-format.test.ts': 1_360_000, 'test/skill-e2e-auto-decide-preserved.test.ts': 1_020_000, 'test/skill-e2e-plan-ceo-finding-floor.test.ts': 1_020_000, 'test/skill-e2e-plan-eng-finding-floor.test.ts': 1_020_000, @@ -33,38 +33,22 @@ const expectedWalls = { 'test/skill-e2e-plan-mode-no-op.test.ts': 3_120_000, 'test/skill-e2e-plan-ceo-mode-routing.test.ts': 1_320_000, 'test/skill-e2e-plan-eng-plan-mode.test.ts': 1_320_000, - 'test/skill-e2e-plan-prosons.test.ts': 2_600_000, + 'test/skill-e2e-plan-prosons.test.ts': 1_360_000, 'test/skill-e2e-plan.test.ts': 3_720_000, }; -test('retry rule: only files whose every case is CAPTURE tier or shorter retry; longer cases run once', () => { - expect(RETRY_MAX_CASE_MS).toBe(ALL_TIERS.CAPTURE_MS + 15_000); +test('paid evals never retry: every paid file and registered row runs once', () => { for (const row of FILE_RETRY_BUDGETS) { - expect(row.retries, row.file).toBe(row.caseMs <= RETRY_MAX_CASE_MS ? (row.file.endsWith('plan-mode-no-op.test.ts') ? 2 : 1) : 0); - expect(retriesForFiles([row.file])).toBe(row.retries); + expect(Object.hasOwn(row, 'retries'), row.file).toBe(false); + expect(retriesForFiles([row.file])).toBe(0); } - expect(FILE_RETRY_BUDGETS.filter(row => row.retries > 0).map(row => row.file).sort()).toEqual([ - 'test/skill-e2e-auq-matrix.test.ts', 'test/skill-e2e-plan-format.test.ts', 'test/skill-e2e-plan-prosons.test.ts', - 'test/skill-e2e-qa-callers.test.ts', 'test/skill-llm-eval.test.ts', - ]); - const paid = collectPaidTestFiles(); - for (const file of SHORT_CASE_RETRY_FILES) { - expect(paid, `stale SHORT_CASE_RETRY_FILES entry: ${file}`).toContain(file); - expect(FILE_RETRY_BUDGETS.some(row => row.file === file)).toBe(false); - const source = read(file); - // Declared short budgets only: a JUDGE/CAPTURE tier or a literal at most the - // cap, no longer tier and no ms literal past the cap. - const literals = [...source.matchAll(/(? Number(match[1]!.replace(/_/g, ''))); - expect(/\b(?:JUDGE_MS|CAPTURE_MS)\b/.test(source) || literals.some(ms => ms >= 60_000 && ms <= RETRY_MAX_CASE_MS), file).toBe(true); - expect(source, file).not.toMatch(/\b(?:CAPTURE_LONG_MS|PTY_MS|PTY_LONG_MS|OVERLAY_CASE_[A-Z_]+)\b/); - expect(literals.filter(ms => ms > RETRY_MAX_CASE_MS && ms < 10_000_000), file).toEqual([]); - expect(retriesForFiles([file])).toBe(1); + for (const file of collectPaidTestFiles()) expect(retriesForFiles([file]), file).toBe(0); + expect(buildPaidShardArgs(['test/x.test.ts'], 1000, 2)).toContain('--retry'); + expect(buildPaidShardArgs(['test/x.test.ts'], 1000, 2).join(' ')).toContain('--retry 0'); + const scripts: Record = JSON.parse(read('package.json')).scripts; + for (const [name, command] of Object.entries(scripts)) { + if (/^test:(?:evals|e2e|gate|periodic)/.test(name)) expect(command, name).not.toMatch(/--retry(?:\s+|=)[1-9]/); } - for (const file of paid.filter(file => !SHORT_CASE_RETRY_FILES.includes(file) && !FILE_RETRY_BUDGETS.some(row => row.file === file))) { - expect(retriesForFiles([file]), file).toBe(0); - } - expect(retriesForFiles([SHORT_CASE_RETRY_FILES[0]!, 'test/skill-e2e-plan.test.ts'])).toBe(0); }); test('registration covers exactly the seventeen demonstrated full-file retry gaps', () => { @@ -133,14 +117,14 @@ for (const row of newBudgets) { outcomes: [{ files: [key], status: 'passed' as const, exitCode: 0, elapsedMs: 1, executedTests: count, skippedTests: 0, budget: resolvePaidShardBudget([key]) }] }]; - test(`${row.file}: full wall and existing retries propagate through planning`, () => { - expect(retriesForFiles([row.file])).toBe(row.retries); + test(`${row.file}: full wall propagates through planning and runs once`, () => { + expect(retriesForFiles([row.file])).toBe(0); expect(resolvePaidShardBudget([row.file])).toEqual({ timeoutMs: expectedWalls[row.file as keyof typeof expectedWalls], source: 'registered', policyId: row.id }); expect(planPaidShards(['test/a.test.ts', row.file, 'test/z.test.ts'], { maxFilesPerShard: 3 })).toContainEqual([row.file]); expect(() => resolvePaidShardBudget([row.file, 'test/neighbor.test.ts'])).toThrow('own shard'); expect(resolvePaidShardBudget([row.file], 50)).toEqual({ timeoutMs: 50, source: 'explicit', policyId: row.id }); expect(buildPaidShardArgs([row.file], row.shardMs, 2, retriesForFiles([row.file]))).toEqual([ - 'test', row.file, '--retry', String(row.retries), '--concurrent', '--max-concurrency=2', `--timeout=${row.shardMs}`, + 'test', row.file, '--retry', '0', '--concurrent', '--max-concurrency=2', `--timeout=${row.shardMs}`, ]); }); @@ -191,17 +175,17 @@ test('quality judge supervision includes the added judge without changing ordina expect(ALL_TIERS).toEqual({ JUDGE_MS: 120000, CAPTURE_MS: 300000, CAPTURE_LONG_MS: 600000, PTY_MS: 900000, PTY_LONG_MS: 1200000 }); const quality = 'test/skill-llm-eval.test.ts'; const qualityBudget = FILE_RETRY_BUDGETS.find(row => row.file === quality)!; - expect(resolvePaidShardBudget([quality])).toEqual({ timeoutMs: 6_220_000, source: 'registered', policyId: qualityBudget.id }); - expect(retriesForFiles([quality])).toBe(1); + expect(resolvePaidShardBudget([quality])).toEqual({ timeoutMs: 3_170_000, source: 'registered', policyId: qualityBudget.id }); + expect(retriesForFiles([quality])).toBe(0); const qualitySource = read(quality); const judgeTimeouts = [...qualitySource.matchAll(/}\s*,\s*(JUDGE_MS|WORKFLOW_JUDGE_TEST_MS)\s*\);/g)].map(match => match[1]); expect(judgeTimeouts.filter(timeout => timeout === 'JUDGE_MS')).toHaveLength(7); expect(judgeTimeouts.filter(timeout => timeout === 'WORKFLOW_JUDGE_TEST_MS')).toHaveLength(17); expect(qualitySource).toContain('WORKFLOW_JUDGE_TEST_MS = JUDGE_MS + 10_000'); expect(qualitySource).toContain('const workDeadline = started + JUDGE_MS'); - expect(qualityBudget.shardMs).toBe((7 * ALL_TIERS.JUDGE_MS + 17 * (ALL_TIERS.JUDGE_MS + 10_000)) * 2 + 120_000); - expect(FINDING_RETRY_BUDGETS.map(row => [row.cases, row.testMs, row.retries, row.shardMs])).toEqual([ - ...Array(2).fill([1, 1500000, 0, 1620000]), + expect(qualityBudget.shardMs).toBe(7 * ALL_TIERS.JUDGE_MS + 17 * (ALL_TIERS.JUDGE_MS + 10_000) + 120_000); + expect(FINDING_RETRY_BUDGETS.map(row => [row.cases, row.testMs, row.shardMs])).toEqual([ + ...Array(2).fill([1, 1500000, 1620000]), ]); for (const tier of ['gate', 'periodic'] as const) { const m = buildRunManifest({ tier, sliceCount: 1, evalsAll: true, env: { EVALS_ALL: '1' } }); @@ -227,7 +211,7 @@ test('detached PR fallback and release commands cover their actual default worke const prFloor = Math.ceil((Math.ceil(fullGateFiles.length / prWorkers) * 1_800_000 + fullGateFiles.reduce( (total, file) => total + Math.max(0, resolvePaidShardBudget([file]).timeoutMs - 1_800_000), 0, )) / 1000 * 1.05); - expect(prFloor).toBe(77_501); + expect(prFloor).toBe(72_755); expect(prWall).toBe(92_820_000); expect(prWall).toBeGreaterThanOrEqual(paidShardWallUpperBoundMs(files, prWorkers) + 120_000); @@ -246,8 +230,8 @@ test('detached PR fallback and release commands cover their actual default worke )) / 1000 * 1.05)); } const detachedReleaseWall = Number(scripts['eval:bg:release'].match(/--timeout (\d+)/)?.[1]) * 1000; - expect(releaseFloors).toEqual([26_471, 30_797]); - expect(releaseFloors.reduce((total, floor) => total + floor, 0)).toBe(57_268); + expect(releaseFloors).toEqual([21_725, 22_481]); + expect(releaseFloors.reduce((total, floor) => total + floor, 0)).toBe(44_206); expect(detachedReleaseWall).toBe(116_700_000); expect(detachedReleaseWall).toBeGreaterThanOrEqual(releaseWall + 120_000); }); @@ -301,7 +285,7 @@ test('both gate executors plan the complete census and supervise every planned s } }); -test('the periodic executor supervises every actual case and retry within its planned CI wall', () => { +test('the periodic executor supervises every actual case within its planned CI wall', () => { const workflow: any = Bun.YAML.parse(read('.github/workflows/evals-periodic.yml')); const executor = workflow.jobs['eval-slices']; const emit = workflow.jobs['plan-slices'].steps.filter((step: any) => @@ -318,7 +302,7 @@ test('the periodic executor supervises every actual case and retry within its pl evalsAll: true, env: { EVALS_ALL: '1' } }); const census = manifest.entries.filter(row => row.status === 'planned'); expect(new Set(census.map(row => shardFile(row.file)))).toEqual(new Set(selectPaidTestFiles(collectPaidTestFiles(), 'periodic').selected)); - expect(census.find(row => row.file === 'test/skill-llm-eval.test.ts')?.budget?.timeoutMs).toBe(6_220_000); + expect(census.find(row => row.file === 'test/skill-llm-eval.test.ts')?.budget?.timeoutMs).toBe(3_170_000); const walls = Array.from({ length: manifest.sliceCount }, (_, i) => sliceSupervisedWallMs(sliceExecutionOrder( census.filter(row => row.slice === i + 1)).map(row => row.file), active.jobs)); expect(manifest.plan!.ciTimeoutMinutes * 60_000).toBeGreaterThanOrEqual(Math.max(...walls) + 20 * 60_000); diff --git a/test/paid-run-manifest.test.ts b/test/paid-run-manifest.test.ts index 9539df224..b91da2547 100644 --- a/test/paid-run-manifest.test.ts +++ b/test/paid-run-manifest.test.ts @@ -487,8 +487,8 @@ describe('hollow-shard guard', () => { }); describe('retry parity', () => { - test('registered native workflows follow the retry rule while overlay attempts stay isolated', () => { - // A 25-minute case is past RETRY_MAX_CASE_MS: its timed-out attempt is the verdict. + test('registered native workflows and overlays run once', () => { + // Paid evals never retry: a timed-out attempt is the verdict. const native = 'test/skill-e2e-plan-ceo-split-overflow.test.ts'; expect(retriesForFiles([native])).toBe(0); expect(retriesForFiles([native.replaceAll('/', '\\')])).toBe(0); @@ -496,16 +496,15 @@ describe('retry parity', () => { const overlay = 'test/skill-e2e-overlay-harness-claude-dedicated-tools-vs-bash.test.ts'; expect(retriesForFiles([overlay])).toBe(0); }); - test('the matrix-era earned retries now follow the timeout-is-a-verdict rule, and each names a real file', () => { - // These three old matrix rows earned `retries: 2`; every one has a - // CAPTURE_LONG case, so a timed-out attempt is now their verdict. + test('the matrix-era earned retries are retired, and each names a real file', () => { + // These three old matrix rows earned `retries: 2`; paid evals never retry. for (const file of ['test/skill-e2e-office-hours-auto-mode.test.ts', 'test/skill-e2e-plan-mode-no-op.test.ts', 'test/skill-e2e-workflow.test.ts']) { expect(fs.existsSync(path.join(ROOT, file)), `stale retry parity entry: ${file}`).toBe(true); expect(retriesForFiles([file])).toBe(0); } expect(retriesForFiles(['test/skill-e2e-retro.test.ts'])).toBe(0); - expect(retriesForFiles(['test/skill-e2e-review.test.ts'])).toBe(1); + expect(retriesForFiles(['test/skill-e2e-review.test.ts'])).toBe(0); expect(buildPaidShardArgs(['x'], 1000, 4, 2)).toContain('2'); - expect(buildPaidShardArgs(['x'], 1000, 4).join(' ')).toContain('--retry 1'); + expect(buildPaidShardArgs(['x'], 1000, 4).join(' ')).toContain('--retry 0'); }); }); diff --git a/test/review-finalization-budget.test.ts b/test/review-finalization-budget.test.ts index 84ed07275..d1eb0f5f4 100644 --- a/test/review-finalization-budget.test.ts +++ b/test/review-finalization-budget.test.ts @@ -1,4 +1,4 @@ -/** The real review registrations must finish capture cleanup before Bun retries. */ +/** The real review registrations record late results and clean up before finalization, under the production zero-retry arguments. */ import { expect, test } from 'bun:test'; import * as fs from 'node:fs'; import * as os from 'node:os'; @@ -12,7 +12,7 @@ const CASES = [ ['review-design-lite', 400, 35], ] as const; for (const [id, workMs, maxTurns] of CASES) { - test.each(['recover', 'both-timeout'])(`${id} records late results before retry or finalization: %s`, scenario => { + test.each(['success', 'timeout'])(`${id} records late results before finalization: %s`, scenario => { const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'review-finalization-')); const script = path.join(dir, 'registration.test.ts'); const facts = path.join(dir, 'events.jsonl'); @@ -41,7 +41,7 @@ mock.module(path.join(root, 'test/helpers/session-runner.ts'), () => ({ runSkillTest: async opts => { const id = ++attempts; event({ kind: 'start', id, timeout: opts.timeout, maxTurns: opts.maxTurns, cwd: opts.workingDirectory }); - const timeout = id === 1 || ${JSON.stringify(scenario)} === 'both-timeout'; + const timeout = ${JSON.stringify(scenario)} === 'timeout'; // Use the caller's actual work budget; only the provider and budget // constants are scaled. The actual registered Bun outer deadline stays. await new Promise(resolve => setTimeout(resolve, timeout ? opts.timeout + 50 : 80)); @@ -62,27 +62,25 @@ await import(path.join(root, ${JSON.stringify(PAID_FILE)})); `); try { const retries = retriesForFiles([PAID_FILE]); - expect(retries).toBe(1); + expect(retries).toBe(0); const child = Bun.spawnSync([process.execPath, ...buildPaidShardArgs([script], resolvePaidShardTimeoutMs([PAID_FILE]), 2, retries)], { cwd: ROOT, timeout: 15_000, stdout: 'pipe', stderr: 'pipe', env: { ...process.env, EVALS: '', EVALS_ALL: '', TMPDIR: dir, TMP: dir, TEMP: dir }, }); const output = child.stdout.toString() + child.stderr.toString(); - expect(child.exitCode, output).toBe(scenario === 'recover' ? 0 : 1); + expect(child.exitCode, output).toBe(scenario === 'success' ? 0 : 1); expect(output).not.toContain('Unhandled error between tests'); const events = fs.readFileSync(facts, 'utf8').trim().split('\n').map(line => JSON.parse(line)); const starts = events.filter(event => event.kind === 'start'); const ready = events.filter(event => event.kind === 'ready'); const records = events.filter(event => event.kind === 'record'); - expect(starts.map(event => event.id)).toEqual([1, 2]); + expect(starts.map(event => event.id)).toEqual([1]); expect(starts.map(({ timeout, maxTurns }) => ({ timeout, maxTurns }))) - .toEqual([{ timeout: workMs, maxTurns }, { timeout: workMs, maxTurns }]); - expect(ready).toEqual([{ kind: 'ready', id: 1, fixtureExists: true }, { kind: 'ready', id: 2, fixtureExists: true }]); + .toEqual([{ timeout: workMs, maxTurns }]); + expect(ready).toEqual([{ kind: 'ready', id: 1, fixtureExists: true }]); expect(records.map(event => [event.id, event.exitReason])) - .toEqual([[1, 'timeout'], [2, scenario === 'recover' ? 'success' : 'timeout']]); + .toEqual([[1, scenario === 'success' ? 'success' : 'timeout']]); expect(events.findIndex(event => event.kind === 'record' && event.id === 1)) - .toBeLessThan(events.findIndex(event => event.kind === 'start' && event.id === 2)); - expect(events.findIndex(event => event.kind === 'record' && event.id === 2)) .toBeLessThan(events.findIndex(event => event.kind === 'finalized')); expect(events.filter(event => event.kind === 'finalized')).toHaveLength(1); expect(events.find(event => event.kind === 'registration')).toEqual({ kind: 'registration', name: id, outerMs: workMs + 50 + 5_000 }); diff --git a/test/ship-hook-actor.test.ts b/test/ship-hook-actor.test.ts index 1f9627de8..f3c307cc5 100644 --- a/test/ship-hook-actor.test.ts +++ b/test/ship-hook-actor.test.ts @@ -13,9 +13,9 @@ import { DEFAULT_SHARD_TIMEOUT_MS, retriesForFiles } from '../scripts/test-paid- const cases: ShipHookCase[] = ['ship-managed-hook-refresh', 'ship-unmanaged-hook-consent', 'ship-local-hook-preservation']; type Fault = 'skip-guard' | 'skip-consent' | 'ask-overwrite' | 'direct-install' | 'read-receipts' | 'edit-policy' | 'tamper-receipts' | 'repeat-question' | 'rate-limit'; -test('whole-file supervision covers every F5 case and the unchanged Bun retry', () => { +test('whole-file supervision covers every F5 case run once', () => { for (const [file, count] of [['test/skill-e2e-ship-hook-refresh.test.ts', 1], ['test/skill-e2e-ship-hook-consent.test.ts', 2]] as const) { - expect(retriesForFiles([file])).toBe(1); + expect(retriesForFiles([file])).toBe(0); expect(count * CAPTURE_MS * (retriesForFiles([file]) + 1) + 120000).toBeLessThanOrEqual(DEFAULT_SHARD_TIMEOUT_MS); } }); diff --git a/test/ship-skip-actor.test.ts b/test/ship-skip-actor.test.ts index 605522fc3..a00c72095 100644 --- a/test/ship-skip-actor.test.ts +++ b/test/ship-skip-actor.test.ts @@ -152,9 +152,9 @@ function protocol(fault?: Fault, billing?: Array, controls: return { provider, directory: () => directory, calls: () => calls, sessions }; } -test('one bounded native case preserves the existing whole-file retry allowance', () => { +test('one bounded native case fits the whole-file wall and never retries', () => { const retries = retriesForFiles(['test/skill-e2e-ship-skip.test.ts']); - expect(retries).toBe(1); + expect(retries).toBe(0); expect(CAPTURE_MS * (retries + 1) + 120000).toBeLessThanOrEqual(DEFAULT_SHARD_TIMEOUT_MS); }); diff --git a/test/workflow-boundaries-fixture.test.ts b/test/workflow-boundaries-fixture.test.ts index f94961a3a..0db8cfb06 100644 --- a/test/workflow-boundaries-fixture.test.ts +++ b/test/workflow-boundaries-fixture.test.ts @@ -292,12 +292,12 @@ test('F9 changed-input selection produces three cases with exact patterns and co expect(prProfileTestNamePattern(files[1], selected.selection)).toBe('(?:^|\\s)(?:investigate-owned-abort|investigate-owned-ending-error)$'); }); -test('both F9 files fit the existing wall with every Bun retry and reserve', () => { +test('both F9 files fit the existing wall with their one run and reserve', () => { for (const file of files) { const source = fs.readFileSync(path.join(import.meta.dir, '..', file), 'utf8'); const count = PR_PROFILE_FILES[file].length; expect([...source.matchAll(/\}, CAPTURE_MS\);/g)]).toHaveLength(count); - expect(retriesForFiles([file])).toBe(1); + expect(retriesForFiles([file])).toBe(0); const budget = resolvePaidShardBudget([file]); expect(budget).toEqual({ timeoutMs: DEFAULT_SHARD_TIMEOUT_MS, source: 'default', policyId: null }); expect(count * CAPTURE_MS * (retriesForFiles([file]) + 1) + 120000).toBeLessThanOrEqual(budget.timeoutMs); From 3f68572cab88d1b13f0849f2cbfc59fdd2a37331 Mon Sep 17 00:00:00 2001 From: garrytan Date: Tue, 29 Sep 2026 19:03:52 +0000 Subject: [PATCH 05/15] test(llm-judge): sample every judge as a pre-registered 3-sample panel Each of the 24 skill-llm-eval judges now draws EVAL_POLICY.judge.samples independent samples of the same prompt concurrently inside the unchanged JUDGE_MS budget. Numeric dimensions gate on the per-dimension panel mean against the unchanged threshold; booleans (would_browse, consistent) on a strict majority. An erroring sample fails the whole panel and is never resampled; a refusal is an unscored panel only when every sample refused. callJudge's 429 backoff stays: it is transport before any model output. The workflow-judge cache stores and validates only complete panels, and its identity now records the panel and zero file retries. Harness tests that pinned one provider call per case now pin the panel size. --- scripts/typecheck-test-baseline.json | 1 - test/cookie-workflow-judge-input.test.ts | 14 +- test/helpers/llm-judge.ts | 59 +++++++ test/helpers/periodic-exclude-data.ts | 8 +- test/helpers/workflow-judge-cache.ts | 35 +++-- test/skill-llm-eval.test.ts | 101 ++++++------ test/workflow-judge-cache.test.ts | 192 +++++++++++++++++------ 7 files changed, 298 insertions(+), 112 deletions(-) diff --git a/scripts/typecheck-test-baseline.json b/scripts/typecheck-test-baseline.json index c309682d1..612942cf7 100644 --- a/scripts/typecheck-test-baseline.json +++ b/scripts/typecheck-test-baseline.json @@ -261,7 +261,6 @@ "test/helpers/shared-libs-eval-fixture.ts\tTS7006\tParameter 'candidate' implicitly has an 'any' type.": 2, "test/helpers/shared-libs-path-fixture.ts\tTS2352\tConversion of type '{ root: string; repo: string; state: string; env: { GSTACK_HOME: string; }; }' to type 'SharedLibsFixture' may be a mistake because neither type sufficiently overlaps with the other. If this was intentional, convert the expression to 'unknown' first. Type '{ root: string; repo: string; state: string; env: { GSTACK_HOME: string; }; }' is missing the following properties from type 'SharedLibsFixture': bin, trace, hookTrace, tip": 1, "test/helpers/shared-libs-plan-actor.ts\tTS18046\t'questions' is of type 'unknown'.": 1, - "test/helpers/workflow-judge-cache.ts\tTS2352\tConversion of type 'string | number | boolean | EvalCacheValue[] | { [key: string]: EvalCacheValue; } | null' to type 'JudgeScore' may be a mistake because neither type sufficiently overlaps with the other. If this was intentional, convert the expression to 'unknown' first. Type '{ [key: string]: EvalCacheValue; }' is missing the following properties from type 'JudgeScore': clarity, completeness, actionability, reasoning": 1, "test/impeccable-fixtures.test.ts\tTS2769\tNo overload matches this call. The last overload gave the following error. Argument of type 'string | undefined' is not assignable to parameter of type 'string'. Type 'undefined' is not assignable to type 'string'.": 1, "test/llm-judge-abort.test.ts\tTS2769\tNo overload matches this call. The last overload gave the following error. Argument of type '{ passed: boolean; }' is not assignable to parameter of type 'undefined'.": 3, "test/llm-judge-frontier.test.ts\tTS2769\tNo overload matches this call. The last overload gave the following error. Argument of type '{ score: number; reason: string; }' is not assignable to parameter of type 'undefined'.": 1, diff --git a/test/cookie-workflow-judge-input.test.ts b/test/cookie-workflow-judge-input.test.ts index cc174995e..70e2b489c 100644 --- a/test/cookie-workflow-judge-input.test.ts +++ b/test/cookie-workflow-judge-input.test.ts @@ -11,7 +11,7 @@ import { selectTests } from './helpers/test-selection'; import { E2E_TOUCHFILES, LLM_JUDGE_TOUCHFILES } from './helpers/touchfiles-data'; import { selectPrProfile } from '../scripts/test-pr-profile'; import { JUDGE_MS } from './helpers/eval-budgets'; -import { JudgeRefusalError, DEFAULT_JUDGE_MAX_TOKENS } from './helpers/llm-judge'; +import { JudgeRefusalError, DEFAULT_JUDGE_MAX_TOKENS, judgePanel, judgePanelMean, judgePanelReasoning, JUDGE_SCORE_DIMENSIONS, JUDGE_PANEL_SAMPLES } from './helpers/llm-judge'; import { COOKIE_MANUAL_REVIEW_FILE, getCookieWorkflowManualReview, isManualReviewEntry } from './helpers/cookie-workflow-manual-review'; const ROOT = resolve(import.meta.dir, '..'); @@ -70,7 +70,7 @@ function actualCookieCallback(root: string, overrides: { const records: EvalTestEntry[] = []; const attempts = new Map(); let callback: () => Promise = async () => { throw new Error('Judge callback was not registered'); }; - new Function('describeIfSelected', 'testIfSelected', 'ROOT', 'buildCookieWorkflowJudgeInput', 'resolveEvalModel', 'callJudge', 'COOKIE_WORKFLOW_JUDGE', 'JUDGE_MS', 'WORKFLOW_JUDGE_TEST_MS', 'WORKFLOW_JUDGE_RECORD_MS', 'evalCollector', 'expect', 'console', 'readWorkflowJudgeInput', 'buildWorkflowJudgePrompt', 'prepareWorkflowJudgeCache', 'workflowJudgeAttempts', 'performance', 'setTimeout', 'clearTimeout', 'JudgeRefusalError', 'getCookieWorkflowManualReview', 'DEFAULT_JUDGE_MAX_TOKENS', 'WORKFLOW_JUDGE_RESPONSE_SCHEMA', 'validWorkflowJudgeScore', registration)( + new Function('describeIfSelected', 'testIfSelected', 'ROOT', 'buildCookieWorkflowJudgeInput', 'resolveEvalModel', 'callJudge', 'COOKIE_WORKFLOW_JUDGE', 'JUDGE_MS', 'WORKFLOW_JUDGE_TEST_MS', 'WORKFLOW_JUDGE_RECORD_MS', 'evalCollector', 'expect', 'console', 'readWorkflowJudgeInput', 'buildWorkflowJudgePrompt', 'prepareWorkflowJudgeCache', 'workflowJudgeAttempts', 'performance', 'setTimeout', 'clearTimeout', 'JudgeRefusalError', 'getCookieWorkflowManualReview', 'DEFAULT_JUDGE_MAX_TOKENS', 'WORKFLOW_JUDGE_RESPONSE_SCHEMA', 'validWorkflowJudgeScore', 'judgePanel', 'judgePanelMean', 'judgePanelReasoning', 'JUDGE_SCORE_DIMENSIONS', registration)( (_suite: string, names: string[], run: () => void) => { expect(names).toEqual([NAME]); run(); }, (name: string, run: () => Promise, budget: number) => { expect(name).toBe(NAME); expect(budget).toBe(JUDGE_MS + 10_000); callback = run; }, root, buildCookieWorkflowJudgeInput, (_kind: string, explicit?: string) => explicit ?? 'fixture-model', @@ -85,7 +85,7 @@ function actualCookieCallback(root: string, overrides: { attempts, overrides.clock ? { now: overrides.clock } : performance, overrides.setTimer ?? setTimeout, overrides.clearTimer ?? clearTimeout, JudgeRefusalError, getCookieWorkflowManualReview, DEFAULT_JUDGE_MAX_TOKENS, - WORKFLOW_JUDGE_RESPONSE_SCHEMA, validWorkflowJudgeScore, + WORKFLOW_JUDGE_RESPONSE_SCHEMA, validWorkflowJudgeScore, judgePanel, judgePanelMean, judgePanelReasoning, JUDGE_SCORE_DIMENSIONS, ); return { run: () => callback(), requests, records, attempts }; } @@ -96,7 +96,7 @@ describe('cookie workflow judge input', () => { approveFixture(root); const h = actualCookieCallback(root, { judge: async () => { throw refusal(); } }); await h.run(); - expect(h.requests).toHaveLength(1); + expect(h.requests).toHaveLength(JUDGE_PANEL_SAMPLES); expect(h.records).toHaveLength(1); expect(h.records[0]).toMatchObject({ passed: false, execution: 'executed', exit_reason: 'provider_refusal' }); expect(isManualReviewEntry(h.records[0])).toBe(true); @@ -161,7 +161,7 @@ describe('cookie workflow judge input', () => { const root = fixture(); approveFixture(root); let calls = 0; const h = actualCookieCallback(root, { judge: async () => { - if (++calls === 1) return { ...passingScore, clarity: 1 }; + if (++calls <= JUDGE_PANEL_SAMPLES) return { ...passingScore, clarity: 1 }; throw refusal(); } }); await expect(h.run()).rejects.toThrow(); @@ -275,7 +275,7 @@ describe('cookie workflow judge input', () => { let scores = passingScore; const h = actualCookieCallback(root, { judge: async () => scores }); await h.run(); - expect(h.requests).toHaveLength(1); + expect(h.requests).toHaveLength(JUDGE_PANEL_SAMPLES); expect(h.requests[0].prompt).toBe(input.prompt); expect(h.requests[0].model).toBe(COOKIE_WORKFLOW_JUDGE.model); expect(h.requests[0].signal).toBeInstanceOf(AbortSignal); @@ -283,7 +283,7 @@ describe('cookie workflow judge input', () => { expect(existsSync(join(root, 'cache'))).toBe(false); const fresh = actualCookieCallback(root); await fresh.run(); - expect(fresh.requests).toHaveLength(1); + expect(fresh.requests).toHaveLength(JUDGE_PANEL_SAMPLES); for (const dimension of ['clarity', 'completeness', 'actionability'] as const) { scores = { ...COOKIE_WORKFLOW_JUDGE.thresholds, [dimension]: COOKIE_WORKFLOW_JUDGE.thresholds[dimension] - 1, reasoning: 'Synthetic failing fixture score' }; await expect(h.run()).rejects.toThrow(); diff --git a/test/helpers/llm-judge.ts b/test/helpers/llm-judge.ts index 4a0f72a1c..da7811b47 100644 --- a/test/helpers/llm-judge.ts +++ b/test/helpers/llm-judge.ts @@ -23,6 +23,8 @@ export interface JudgeScore { reasoning: string; } +export const JUDGE_SCORE_DIMENSIONS = ['clarity', 'completeness', 'actionability'] as const; + export interface JudgeRefusalEvidence { stop_reason: 'refusal'; response_id: string | null; @@ -196,6 +198,63 @@ export async function callJudge( } } +/** + * Samples per judge panel: EVAL_POLICY.judge.samples, restated here so this + * helper (imported by many paid tests) does not pull the quarantine registry + * into their touchfile closure. test/judge-panel.test.ts pins the two equal. + */ +export const JUDGE_PANEL_SAMPLES = 3; + +/** + * Judge panel (EVAL_POLICY.judge): every `judge`-kind entry draws a fixed number of + * independent samples of the SAME prompt concurrently, inside its unchanged + * JUDGE_MS budget. Numeric dimensions gate on the per-dimension panel mean + * against the unchanged minimum; boolean fields gate on a strict majority. + * A sample that errors (refusal, truncation, non-JSON, malformed field) fails + * the whole panel and is never resampled. callJudge's 429 backoff happens + * before any model output exists, so it is transport, not a verdict retry. + */ +export async function judgePanel(sample: () => Promise): Promise { + const settled = await Promise.allSettled(Array.from({ length: JUDGE_PANEL_SAMPLES }, () => sample())); + const failures = settled.flatMap((result, index) => result.status === 'rejected' ? [{ index, reason: result.reason }] : []); + if (failures.length === 0) return settled.map(result => (result as PromiseFulfilledResult).value); + const first = failures[0]!; + // A refusal is an unscored panel only when EVERY sample refused; a partial + // refusal beside scored samples is an ordinary failed panel. + if (first.reason instanceof JudgeRefusalError && failures.length < settled.length) { + throw new Error(`Judge panel sample ${first.index + 1} of ${settled.length} failed beside scored samples: ${first.reason.message}`); + } + throw first.reason; +} + +/** Per-dimension mean over a panel; any non-finite sample value fails the panel. */ +export function judgePanelMean(samples: ReadonlyArray>, keys: readonly K[]): Record { + if (samples.length === 0) throw new Error('Judge panel has no samples'); + return Object.fromEntries(keys.map(key => { + const values = samples.map(sample => sample && typeof sample === 'object' ? sample[key] : undefined); + const bad = values.findIndex(value => typeof value !== 'number' || !Number.isFinite(value)); + if (bad !== -1) throw new Error(`Judge panel sample ${bad + 1} has non-numeric ${key}: ${JSON.stringify(values[bad])}`); + return [key, (values as number[]).reduce((sum, value) => sum + value, 0) / values.length]; + })) as Record; +} + +/** Strict majority of a boolean field; any non-boolean sample value fails the panel. */ +export function judgePanelMajority(samples: ReadonlyArray>, key: K): boolean { + if (samples.length === 0) throw new Error('Judge panel has no samples'); + const values = samples.map(sample => sample && typeof sample === 'object' ? sample[key] : undefined); + const bad = values.findIndex(value => typeof value !== 'boolean'); + if (bad !== -1) throw new Error(`Judge panel sample ${bad + 1} has non-boolean ${key}: ${JSON.stringify(values[bad])}`); + return values.filter(value => value === true).length * 2 > values.length; +} + +/** Sample reasoning lines, numbered, for the collector record. */ +export function judgePanelReasoning(samples: ReadonlyArray): string { + return samples.map((sample, index) => { + const reasoning = sample && typeof sample === 'object' ? (sample as { reasoning?: unknown }).reasoning : undefined; + return `[sample ${index + 1}] ${typeof reasoning === 'string' ? reasoning : ''}`; + }).join('\n'); +} + /** * Score documentation quality on clarity/completeness/actionability (1-5). */ diff --git a/test/helpers/periodic-exclude-data.ts b/test/helpers/periodic-exclude-data.ts index 5959f44d7..5628e13d7 100644 --- a/test/helpers/periodic-exclude-data.ts +++ b/test/helpers/periodic-exclude-data.ts @@ -72,7 +72,12 @@ export const CASE_CI_EXCLUDE: Record= `exit.rate` over >= * `exit.minTrials`; at most `capFraction` of each tier's * blocking cases; an entry expires after `expiryWeeklyRuns`. - * drift - one-sided Fisher exact alarm between input-identity series. + * judge - a judge case draws `samples` independent samples of one + * prompt concurrently; numeric dimensions gate on the panel + * mean against the unchanged threshold, booleans on a strict + * majority; an erroring sample fails the panel, never resampled. + * drift - one-sided Fisher exact alarm between input-identity series + * (Holm-controlled across the cases tested in one report). * infraRedispatch - a census whose every red verdict is machine-classified * INFRA or INCOMPLETE may be re-dispatched this many times as * a new run; both runs are reported. @@ -86,6 +91,7 @@ export const EVAL_POLICY = { capFraction: 0.10, expiryWeeklyRuns: 8, }, + judge: { samples: 3 }, drift: { fisherAlpha: 0.05, fisherMinPerSide: 6 }, infraRedispatch: 1, } as const; diff --git a/test/helpers/workflow-judge-cache.ts b/test/helpers/workflow-judge-cache.ts index 0b421567a..ee9607546 100644 --- a/test/helpers/workflow-judge-cache.ts +++ b/test/helpers/workflow-judge-cache.ts @@ -4,7 +4,7 @@ import * as path from 'node:path'; import { spawnSync } from 'node:child_process'; import { DEFAULT_JUDGE_MAX_TOKENS, resolveEvalModel } from '../../lib/eval-model'; import { JUDGE_MS } from './eval-budgets'; -import type { JudgeScore } from './llm-judge'; +import { JUDGE_PANEL_SAMPLES, JUDGE_SCORE_DIMENSIONS, judgePanelMean, type JudgeScore } from './llm-judge'; import { readWorkflowJudgeInput, buildWorkflowJudgePrompt, WORKFLOW_JUDGE_RESPONSE_SCHEMA, WORKFLOW_JUDGE_REASONING_WORD_LIMIT } from './workflow-judge-input'; import { buildEvalInputIdentity, lookupEvalInputCache, sourceDependencyClosure, storeEvalInputCache, type EvalCacheValue, type EvalInputIdentity, type EvalPassingProof } from '../../scripts/eval-input-cache'; @@ -39,17 +39,28 @@ export function validWorkflowJudgeScore(value: EvalCacheValue, thresholds: Thres || typeof value.reasoning !== 'string' || (structuredResponse && (!value.reasoning.trim() || value.reasoning.trim().split(/\s+/).length >= WORKFLOW_JUDGE_REASONING_WORD_LIMIT))) return false; - return (['clarity', 'completeness', 'actionability'] as const).every(key => + return JUDGE_SCORE_DIMENSIONS.every(key => typeof value[key] === 'number' && Number.isInteger(value[key]) && value[key] >= thresholds[key] && value[key] <= 5); } +const SAMPLE_RANGE: Thresholds = { clarity: 1, completeness: 1, actionability: 1 }; + +/** A complete judge panel: exactly JUDGE_PANEL_SAMPLES of valid samples whose per-dimension mean meets every threshold. */ +export function validWorkflowJudgePanel(value: EvalCacheValue, thresholds: Thresholds, structuredResponse = false): value is { samples: Array } { + if (!value || typeof value !== 'object' || Array.isArray(value) || Object.keys(value).join(',') !== 'samples' + || !Array.isArray(value.samples) || value.samples.length !== JUDGE_PANEL_SAMPLES + || !value.samples.every(sample => validWorkflowJudgeScore(sample, SAMPLE_RANGE, structuredResponse))) return false; + const mean = judgePanelMean(value.samples as JudgeScore[], JUDGE_SCORE_DIMENSIONS); + return JUDGE_SCORE_DIMENSIONS.every(key => mean[key] >= thresholds[key]); +} + export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): { - lookup(): { scores: JudgeScore; reuse: WorkflowJudgeReuse } | null; + lookup(): { samples: JudgeScore[]; reuse: WorkflowJudgeReuse } | null; /** The attempt guard is rechecked after synchronous input/provenance reads. */ - publish(scores: JudgeScore, isActive?: () => boolean): (() => void) | undefined; + publish(samples: JudgeScore[], isActive?: () => boolean): (() => void) | undefined; } { const env = opts.env ?? process.env; - const noCache = { lookup: () => null, publish: (_scores: JudgeScore) => undefined }; + const noCache = { lookup: () => null, publish: (_samples: JudgeScore[]) => undefined }; const pr = Number(env.EVALS_CACHE_PR); // Runtime ID is the immutable CI image manifest, not a mutable image tag. // Nonstandard Node/Bun preload code or custom model endpoints need a separate @@ -74,7 +85,8 @@ export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): { files: workflowJudgeDependencies(opts.root, input.files.map(file => file.path)), prompts: { [opts.testName]: prompt }, parameters: { rootPackage, thresholds: opts.thresholds, max_tokens: opts.maxTokens ?? DEFAULT_JUDGE_MAX_TOKENS, temperature: null, budget_ms: JUDGE_MS, - request: opts.stream ? 'messages.stream/user' : 'messages.create/user', retries: 1, + request: opts.stream ? 'messages.stream/user' : 'messages.create/user', retries: 0, + panel: { samples: JUDGE_PANEL_SAMPLES, numeric: 'mean', boolean: 'majority' }, ...(opts.stream ? { stream: true } : {}), ...(opts.structuredResponse ? { output_config: { format: { type: 'json_schema', schema: WORKFLOW_JUDGE_RESPONSE_SCHEMA } }, response_validation: { reasoning_words_below: WORKFLOW_JUDGE_REASONING_WORD_LIMIT } } : {}) }, @@ -94,14 +106,15 @@ export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): { return { lookup() { const result = lookupEvalInputCache({ ...common, identity: before, - validateResult: value => validWorkflowJudgeScore(value, opts.thresholds, opts.structuredResponse) }); + validateResult: value => validWorkflowJudgePanel(value, opts.thresholds, opts.structuredResponse) }); return result.status === 'reused' - ? { scores: result.result as JudgeScore, reuse: { key: result.key, source: result.source } } : null; + ? { samples: (result.result as unknown as { samples: JudgeScore[] }).samples, reuse: { key: result.key, source: result.source } } : null; }, - publish(scores, isActive = () => true) { + publish(samples, isActive = () => true) { // Caller reaches here ONLY after its actual assertions passed. A later // failed case in the file does not erase this independently completed case. - if (!isActive() || !validWorkflowJudgeScore(scores as unknown as EvalCacheValue, opts.thresholds, opts.structuredResponse)) return; + const panel = { samples: samples.map(({ clarity, completeness, actionability, reasoning }) => ({ clarity, completeness, actionability, reasoning })) }; + if (!isActive() || !validWorkflowJudgePanel(panel as unknown as EvalCacheValue, opts.thresholds, opts.structuredResponse)) return; const after = currentIdentity(); const runId = env.GITHUB_RUN_ID ? `${env.GITHUB_RUN_ID}/${env.GITHUB_RUN_ATTEMPT ?? '1'}` : env.EVALS_RUN_ID; if (!after || !runId || !isActive()) return; @@ -112,7 +125,7 @@ export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): { cancelled: false, skipped: 0, failed: 0, passed: 1, cases: [{ id: opts.testName, outcome: 'passed', attempt: 1 }], source: { runId, revision: revision.stdout.trim(), completedAt: Date.now() }, - result: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability, reasoning: scores.reasoning }, + result: panel, } }); // A slow synchronous write can consume the recording allowance. The // caller withdraws this new receipt if its final deadline check fails. diff --git a/test/skill-llm-eval.test.ts b/test/skill-llm-eval.test.ts index 41b5ab13c..af9d1774c 100644 --- a/test/skill-llm-eval.test.ts +++ b/test/skill-llm-eval.test.ts @@ -14,7 +14,7 @@ import { afterAll, expect } from 'bun:test'; import { JUDGE_MS } from './helpers/eval-budgets'; import * as fs from 'fs'; import * as path from 'path'; -import { callJudge, judge, JudgeRefusalError, DEFAULT_JUDGE_MAX_TOKENS } from './helpers/llm-judge'; +import { callJudge, judge, JudgeRefusalError, DEFAULT_JUDGE_MAX_TOKENS, judgePanel, judgePanelMean, judgePanelMajority, judgePanelReasoning, JUDGE_SCORE_DIMENSIONS } from './helpers/llm-judge'; import { ENG_REVIEW_EXCERPT } from './helpers/workflow-excerpt'; import type { JudgeScore } from './helpers/llm-judge'; import { readWorkflowJudgeInput, buildWorkflowJudgePrompt, QA_DISCOVERY_REFERENCES, WORKFLOW_JUDGE_RESPONSE_SCHEMA, type WorkflowJudgeInput } from './helpers/workflow-judge-input'; @@ -100,8 +100,9 @@ describeIfSelected('LLM-as-judge quality evals', [ // rewrites the pin). const section = sliceBrowseSection('## Snapshot Flags'); - const scores = await judge('browse skill reference (flags + commands)', section); - console.log('Browse SKILL.md scores:', JSON.stringify(scores, null, 2)); + const samples = await judgePanel(() => judge('browse skill reference (flags + commands)', section)); + const scores = judgePanelMean(samples, JUDGE_SCORE_DIMENSIONS); + console.log('Browse SKILL.md panel:', JSON.stringify({ mean: scores, samples }, null, 2)); const baselinesPath = path.join(ROOT, 'test', 'fixtures', 'eval-baselines.json'); const baselines = JSON.parse(fs.readFileSync(baselinesPath, 'utf-8')); @@ -120,9 +121,9 @@ describeIfSelected('LLM-as-judge quality evals', [ tier: 'llm-judge', passed: scores.clarity >= 3 && scores.completeness >= 4 && scores.actionability >= 4 && regressions.length === 0, duration_ms: Date.now() - t0, - cost_usd: 0.02, + cost_usd: 0.02 * samples.length, judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability }, - judge_reasoning: regressions.length ? `${scores.reasoning} | ${regressions.join('; ')}` : scores.reasoning, + judge_reasoning: regressions.length ? `${judgePanelReasoning(samples)} | ${regressions.join('; ')}` : judgePanelReasoning(samples), }); expect(scores.clarity).toBeGreaterThanOrEqual(3); @@ -144,8 +145,9 @@ describeIfSelected('LLM-as-judge quality evals', [ if (setupStart < 0 || setupEnd < 0) throw new Error('browse/SKILL.md: setup block not found — regenerate with: bun run gen:skill-docs'); const section = content.slice(setupStart, setupEnd); - const scores = await judge('setup/binary discovery instructions', section); - console.log('Setup block scores:', JSON.stringify(scores, null, 2)); + const samples = await judgePanel(() => judge('setup/binary discovery instructions', section)); + const scores = judgePanelMean(samples, JUDGE_SCORE_DIMENSIONS); + console.log('Setup block panel:', JSON.stringify({ mean: scores, samples }, null, 2)); evalCollector?.addTest({ name: 'setup block', @@ -153,9 +155,9 @@ describeIfSelected('LLM-as-judge quality evals', [ tier: 'llm-judge', passed: scores.actionability >= 3 && scores.clarity >= 3, duration_ms: Date.now() - t0, - cost_usd: 0.02, + cost_usd: 0.02 * samples.length, judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability }, - judge_reasoning: scores.reasoning, + judge_reasoning: judgePanelReasoning(samples), }); // Setup block is intentionally minimal (binary discovery only). @@ -203,7 +205,7 @@ describeIfSelected('QA skill quality evals', ['qa/SKILL.md workflow', 'qa/SKILL. startMarker: '# /qa: Test', endMarker: null, references: ['qa/templates/functional-report-template.md'] }).text; - const scores = await callJudge(`You are evaluating the quality of a QA testing workflow document for an AI coding agent. + const samples = await judgePanel(() => callJudge(`You are evaluating the quality of a QA testing workflow document for an AI coding agent. The agent reads this source-file bundle to select browser, native functional or mixed surfaces, explore with bounded probes, reproduce and diagnose defects, add a regression @@ -222,8 +224,9 @@ Respond with ONLY valid JSON: Here is the QA workflow to evaluate: -${section}`); - console.log('QA workflow scores:', JSON.stringify(scores, null, 2)); +${section}`)); + const scores = judgePanelMean(samples, JUDGE_SCORE_DIMENSIONS); + console.log('QA workflow panel:', JSON.stringify({ mean: scores, samples }, null, 2)); evalCollector?.addTest({ name: 'qa/SKILL.md workflow', @@ -231,9 +234,9 @@ ${section}`); tier: 'llm-judge', passed: scores.clarity >= 3 && scores.completeness >= 3 && scores.actionability >= 4, duration_ms: Date.now() - t0, - cost_usd: 0.02, + cost_usd: 0.02 * samples.length, judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability }, - judge_reasoning: scores.reasoning, + judge_reasoning: judgePanelReasoning(samples), }); expect(scores.clarity).toBeGreaterThanOrEqual(3); @@ -247,7 +250,7 @@ ${section}`); const t0 = Date.now(); const section = sliceQaPatterns('## Health Score Rubric'); - const scores = await callJudge(`You are evaluating a health score rubric that an AI agent must follow to compute a numeric QA score. + const samples = await judgePanel(() => callJudge(`You are evaluating a health score rubric that an AI agent must follow to compute a numeric QA score. The agent uses this rubric after QA testing a website. It needs to: 1. Understand each scoring category and what counts as a deduction @@ -264,8 +267,9 @@ Respond with ONLY valid JSON: Here is the rubric to evaluate: -${section}`); - console.log('QA health rubric scores:', JSON.stringify(scores, null, 2)); +${section}`)); + const scores = judgePanelMean(samples, JUDGE_SCORE_DIMENSIONS); + console.log('QA health rubric panel:', JSON.stringify({ mean: scores, samples }, null, 2)); evalCollector?.addTest({ name: 'qa/SKILL.md health rubric', @@ -273,9 +277,9 @@ ${section}`); tier: 'llm-judge', passed: scores.clarity >= 3 && scores.completeness >= 3 && scores.actionability >= 4, duration_ms: Date.now() - t0, - cost_usd: 0.02, + cost_usd: 0.02 * samples.length, judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability }, - judge_reasoning: scores.reasoning, + judge_reasoning: judgePanelReasoning(samples), }); expect(scores.clarity).toBeGreaterThanOrEqual(3); @@ -294,7 +298,7 @@ ${section}`); const diffAwareSection = sliceQaPatterns('### Diff-aware', '### Full'); const rulesSection = sliceQaPatterns('## Important Rules'); - const result = await callJudge<{ would_browse: boolean; fallback_behavior: string; confidence: number; reasoning: string }>(`You are evaluating whether a QA testing skill document would cause an AI agent to USE THE BROWSER or REFUSE to use the browser in a specific scenario. + const samples = await judgePanel(() => callJudge<{ would_browse: boolean; fallback_behavior: string; confidence: number; reasoning: string }>(`You are evaluating whether a QA testing skill document would cause an AI agent to USE THE BROWSER or REFUSE to use the browser in a specific scenario. SCENARIO: A user runs /qa (a browser-based QA testing skill). The branch diff shows ONLY prompt template files and config file changes — no routes, views, controllers, components, or CSS were changed. The changes are "purely backend" with no obvious UI surface. @@ -318,9 +322,10 @@ Respond with ONLY valid JSON: Rules: - would_browse should be true if the document instructs the agent to always use the browser regardless of diff content - would_browse should be false if the document allows the agent to skip browser testing for non-UI changes -- confidence: 5 = document is unambiguous, 1 = document is unclear or contradictory`); +- confidence: 5 = document is unambiguous, 1 = document is unclear or contradictory`)); + const result = { would_browse: judgePanelMajority(samples, 'would_browse'), ...judgePanelMean(samples, ['confidence'] as const) }; - console.log('QA anti-refusal result:', JSON.stringify(result, null, 2)); + console.log('QA anti-refusal panel:', JSON.stringify({ result, samples }, null, 2)); evalCollector?.addTest({ name: 'qa/SKILL.md anti-refusal', @@ -328,9 +333,9 @@ Rules: tier: 'llm-judge', passed: result.would_browse === true && result.confidence >= 4, duration_ms: Date.now() - t0, - cost_usd: 0.02, + cost_usd: 0.02 * samples.length, judge_scores: { would_browse: result.would_browse ? 1 : 0, confidence: result.confidence }, - judge_reasoning: result.reasoning, + judge_reasoning: judgePanelReasoning(samples), }); expect(result.would_browse).toBe(true); @@ -362,7 +367,7 @@ describeIfSelected('Cross-skill consistency evals', ['cross-skill greptile consi extractGrepLines(retroContent, 'retro/SKILL.md'), ].join('\n\n'); - const result = await callJudge<{ consistent: boolean; issues: string[]; score: number; reasoning: string }>(`You are evaluating whether multiple skill configuration files implement the same data architecture consistently. + const samples = await judgePanel(() => callJudge<{ consistent: boolean; issues: string[]; score: number; reasoning: string }>(`You are evaluating whether multiple skill configuration files implement the same data architecture consistently. INTENDED ARCHITECTURE: - greptile-history has TWO paths: per-project (~/.gstack/projects/{slug}/greptile-history.md) and global (~/.gstack/greptile-history.md) @@ -383,9 +388,10 @@ Evaluate consistency. Respond with ONLY valid JSON: "reasoning": "brief explanation" } -score (1-5): 5 = perfectly consistent, 1 = contradictory`); +score (1-5): 5 = perfectly consistent, 1 = contradictory`)); + const result = { consistent: judgePanelMajority(samples, 'consistent'), ...judgePanelMean(samples, ['score'] as const) }; - console.log('Cross-skill consistency:', JSON.stringify(result, null, 2)); + console.log('Cross-skill consistency panel:', JSON.stringify({ result, samples }, null, 2)); evalCollector?.addTest({ name: 'cross-skill greptile consistency', @@ -393,9 +399,9 @@ score (1-5): 5 = perfectly consistent, 1 = contradictory`); tier: 'llm-judge', passed: result.consistent && result.score >= 4, duration_ms: Date.now() - t0, - cost_usd: 0.02, + cost_usd: 0.02 * samples.length, judge_scores: { consistency_score: result.score }, - judge_reasoning: result.reasoning, + judge_reasoning: judgePanelReasoning(samples), }); expect(result.consistent).toBe(true); @@ -439,7 +445,8 @@ async function runWorkflowJudge(opts: { const workDeadline = started + JUDGE_MS; let stage: 'input' | 'judge' | 'validation' | 'recording' = 'input'; let finalized = false; - let scores: JudgeScore | undefined; + let samples: JudgeScore[] | undefined; + let scores: Record | undefined; let manualReview: ManualJudgeReview | undefined; let customInputMetadata: { prompt: string; model: string } | undefined; let reused: ReturnType['lookup']> = null; @@ -458,19 +465,19 @@ async function runWorkflowJudge(opts: { evalCollector?.addTest({ name: opts.testName, suite: opts.suite, tier: 'llm-judge', passed, attempt, duration_ms: Math.max(0, performance.now() - started), - cost_usd: reused || !scores ? 0 : 0.02, + cost_usd: reused || !samples ? 0 : 0.02 * samples.length, execution: reused ? 'reused' : 'executed', ...customInputMetadata, ...(manualReview ? { manual_review: manualReview } : {}), ...(reused ? { reused_from: { input_key: reused.reuse.key, run_id: reused.reuse.source.runId, revision: reused.reuse.source.revision, completed_at: new Date(reused.reuse.source.completedAt).toISOString() } } : {}), - ...(scores ? { judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability }, - judge_reasoning: scores.reasoning } : {}), + ...(scores ? { judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability } } : {}), + ...(samples ? { judge_reasoning: judgePanelReasoning(samples) } : {}), ...(passed ? {} : { exit_reason: error instanceof JudgeRefusalError ? 'provider_refusal' : error instanceof Error && error.name === 'WorkflowJudgeDeadline' ? 'timeout' : error instanceof Error && error.name === 'WorkflowJudgeSuperseded' ? 'cancelled' : stage === 'validation' ? 'validation_failed' : 'harness_error', - error: `${error instanceof Error ? error.message : String(error)}${scores ? '' : error instanceof JudgeRefusalError + error: `${error instanceof Error ? error.message : String(error)}${samples ? '' : error instanceof JudgeRefusalError ? '\nNo automated score; provider refusal usage retained when manually accepted; cost unavailable.' : '\nNo completed model response; cost and usage unavailable.'}` }), }); @@ -508,11 +515,11 @@ async function runWorkflowJudge(opts: { checkActive(); stage = 'judge'; const maxTokens = opts.maxTokens ?? DEFAULT_JUDGE_MAX_TOKENS; - let result: JudgeScore; + let result: JudgeScore[]; try { - result = reused?.scores ?? await callJudge(prompt, opts.model, { signal: controller.signal, max_tokens: maxTokens, + result = reused?.samples ?? await judgePanel(() => callJudge(prompt, opts.model, { signal: controller.signal, max_tokens: maxTokens, ...(opts.stream ? { stream: true } : {}), - ...(opts.structuredResponse ? { jsonSchema: WORKFLOW_JUDGE_RESPONSE_SCHEMA } : {}) }); + ...(opts.structuredResponse ? { jsonSchema: WORKFLOW_JUDGE_RESPONSE_SCHEMA } : {}) })); } catch (error) { checkActive(); if (error instanceof JudgeRefusalError && customInputMetadata) { @@ -529,20 +536,21 @@ async function runWorkflowJudge(opts: { throw error; } checkActive(); - scores = result; + samples = result; console.log(`[workflow-judge] ${opts.testName}: ${reused ? `reused ${reused.reuse.source.runId} @ ${reused.reuse.source.revision} (${new Date(reused.reuse.source.completedAt).toISOString()})` : 'executed'}`); - console.log(`${opts.testName} scores:`, JSON.stringify(scores, null, 2)); stage = 'validation'; - if (opts.structuredResponse && !validWorkflowJudgeScore(scores as unknown as EvalCacheValue, { clarity: 1, completeness: 1, actionability: 1 }, true)) { + if (opts.structuredResponse && !samples.every(sample => validWorkflowJudgeScore(sample as unknown as EvalCacheValue, { clarity: 1, completeness: 1, actionability: 1 }, true))) { throw new Error('Structured workflow judge violated the response schema'); } + scores = judgePanelMean(samples, JUDGE_SCORE_DIMENSIONS); + console.log(`${opts.testName} panel:`, JSON.stringify({ mean: scores, samples }, null, 2)); expect(scores.clarity).toBeGreaterThanOrEqual(thresholds.clarity); expect(scores.completeness).toBeGreaterThanOrEqual(thresholds.completeness); expect(scores.actionability).toBeGreaterThanOrEqual(thresholds.actionability); checkActive(); stage = 'recording'; arm(); - const discardReceipt = reused ? undefined : cache.publish(scores, active); + const discardReceipt = reused ? undefined : cache.publish(samples, active); try { checkActive(); finish(true); } catch (error) { discardReceipt?.(); throw error; } }; @@ -792,7 +800,7 @@ describeIfSelected('Voice directive eval', ['voice directive tone'], () => { const voiceEnd = content.indexOf('\n## ', voiceStart + 1); const voiceSection = content.slice(voiceStart, voiceEnd > 0 ? voiceEnd : voiceStart + 3000); - const result = await callJudge<{ + const samples = await judgePanel(() => callJudge<{ directness: number; concreteness: number; avoids_corporate: number; @@ -812,9 +820,10 @@ Return JSON only: {"directness": N, "concreteness": N, "avoids_corporate": N, "avoids_ai_vocabulary": N, "connects_user_outcomes": N, "reasoning": "..."} THE VOICE DIRECTIVE: -${voiceSection}`); +${voiceSection}`)); + const result = judgePanelMean(samples, ['directness', 'concreteness', 'avoids_corporate', 'avoids_ai_vocabulary', 'connects_user_outcomes'] as const); - console.log('Voice directive scores:', JSON.stringify(result, null, 2)); + console.log('Voice directive panel:', JSON.stringify({ mean: result, samples }, null, 2)); evalCollector?.addTest({ name: 'voice directive tone', @@ -823,7 +832,7 @@ ${voiceSection}`); passed: result.directness >= 4 && result.concreteness >= 4 && result.avoids_corporate >= 4 && result.avoids_ai_vocabulary >= 4 && result.connects_user_outcomes >= 4, duration_ms: Date.now() - t0, - cost_usd: 0.02, + cost_usd: 0.02 * samples.length, judge_scores: { directness: result.directness, concreteness: result.concreteness, @@ -831,7 +840,7 @@ ${voiceSection}`); avoids_ai_vocabulary: result.avoids_ai_vocabulary, connects_user_outcomes: result.connects_user_outcomes, }, - judge_reasoning: result.reasoning, + judge_reasoning: judgePanelReasoning(samples), }); expect(result.directness).toBeGreaterThanOrEqual(4); diff --git a/test/workflow-judge-cache.test.ts b/test/workflow-judge-cache.test.ts index 79feebcdd..1d55fad33 100644 --- a/test/workflow-judge-cache.test.ts +++ b/test/workflow-judge-cache.test.ts @@ -1,18 +1,22 @@ -import { afterEach, expect, spyOn, test } from 'bun:test'; +import { afterEach, describe, expect, spyOn, test } from 'bun:test'; import { Messages } from '@anthropic-ai/sdk/resources/messages'; -import { callJudge, JudgeRefusalError, DEFAULT_JUDGE_MAX_TOKENS } from './helpers/llm-judge'; +import { callJudge, JudgeRefusalError, DEFAULT_JUDGE_MAX_TOKENS, judgePanel, judgePanelMajority, judgePanelMean, judgePanelReasoning, JUDGE_SCORE_DIMENSIONS, JUDGE_PANEL_SAMPLES } from './helpers/llm-judge'; +import { EVAL_POLICY } from './helpers/periodic-exclude-data'; import { getCookieWorkflowManualReview } from './helpers/cookie-workflow-manual-review'; import { resolveEvalModel } from '../lib/eval-model'; import * as fs from 'node:fs'; import * as os from 'node:os'; import * as path from 'node:path'; import { execFileSync } from 'node:child_process'; -import { prepareWorkflowJudgeCache, validWorkflowJudgeScore, workflowJudgeDependencies, type WorkflowCacheOptions } from './helpers/workflow-judge-cache'; +import { prepareWorkflowJudgeCache, validWorkflowJudgePanel, validWorkflowJudgeScore, workflowJudgeDependencies, type WorkflowCacheOptions } from './helpers/workflow-judge-cache'; import { readWorkflowJudgeInput, buildWorkflowJudgePrompt, QA_DISCOVERY_REFERENCES, WORKFLOW_JUDGE_RESPONSE_SCHEMA } from './helpers/workflow-judge-input'; const roots: string[] = []; afterEach(() => { for (const root of roots.splice(0)) fs.rmSync(root, { recursive: true, force: true }); }); const scores = { clarity: 4, completeness: 5, actionability: 4, reasoning: 'Concrete steps' }; +const SAMPLES = JUDGE_PANEL_SAMPLES; +const panelOf = (sample: typeof scores) => Array.from({ length: SAMPLES }, () => sample); +const panel = panelOf(scores); function fixture() { const root = fs.mkdtempSync(path.join(os.tmpdir(), 'gstack-judge-cache-')); roots.push(root); const files = { @@ -53,9 +57,9 @@ function fixture() { } test('the audited adapter reuses only the exact completed score and original provenance', () => { - const f = fixture(); const first = f.cache(); expect(first.lookup()).toBeNull(); first.publish(scores); + const f = fixture(); const first = f.cache(); expect(first.lookup()).toBeNull(); first.publish(panel); expect(f.entries()).toHaveLength(1); - const reused = f.cache().lookup(); expect(reused?.scores).toEqual(scores); + const reused = f.cache().lookup(); expect(reused?.samples).toEqual(panel); expect(reused?.reuse.source.runId).toBe('free-cache-test'); expect(reused?.reuse.source.revision).toMatch(/^[a-f0-9]{40}$/); expect(reused?.reuse.source.completedAt).toBeLessThanOrEqual(Date.now()); @@ -73,11 +77,11 @@ test('the dependency closure includes actual installed SDK bytes and local trans }); test('release-label changes preserve reuse; other package semantics invalidate it', () => { - const f = fixture(); f.cache().publish(scores); + const f = fixture(); f.cache().publish(panel); const file = path.join(f.root, 'package.json'); const original = JSON.parse(fs.readFileSync(file, 'utf8')); fs.writeFileSync(file, JSON.stringify({ ...original, version: '2.0.0' }, null, 2)); - expect(f.cache().lookup()?.scores).toEqual(scores); + expect(f.cache().lookup()?.samples).toEqual(panel); for (const change of [{ scripts: { 'test:gate': 'changed command' } }, { dependencies: { 'some-sdk': '2.0.0' } }]) { fs.writeFileSync(file, JSON.stringify({ ...original, ...change, version: '2.0.0' })); expect(f.cache().lookup()).toBeNull(); @@ -89,7 +93,7 @@ for (const file of ['test/helpers/nested.ts', 'node_modules/@anthropic-ai/sdk/in 'scripts/test-paid-shards.ts', 'scripts/test-strict-output.ts', 'scripts/eval-select.ts', 'scripts/test-pr-profile.ts', '.github/workflows/evals.yml']) { test(`changes in ${file} require new evaluation`, () => { - const f = fixture(); f.cache().publish(scores); const target = path.join(f.root, file); + const f = fixture(); f.cache().publish(panel); const target = path.join(f.root, file); fs.appendFileSync(target, file.endsWith('.json') ? ' ' : '\n// changed'); f.refreshPrompt(); expect(f.cache().lookup()).toBeNull(); }); @@ -98,8 +102,8 @@ for (const file of ['test/helpers/nested.ts', 'node_modules/@anthropic-ai/sdk/in test('changed sources during an attempt and mismatched actual prompt cannot publish', () => { const f = fixture(); const before = f.cache(); fs.appendFileSync(path.join(f.root, 'example/sections/review.md'), 'new finding'); - before.publish(scores); expect(f.entries()).toHaveLength(0); - f.refreshPrompt(); f.opts.prompt += ' hidden new request'; f.cache().publish(scores); + before.publish(panel); expect(f.entries()).toHaveLength(0); + f.refreshPrompt(); f.opts.prompt += ' hidden new request'; f.cache().publish(panel); expect(f.entries()).toHaveLength(0); }); @@ -108,52 +112,52 @@ for (const [key, value] of Object.entries({ EVALS_FRESH: '1', EVALS_TIER: 'perio EVALS_CACHE_REPOSITORY: '', NODE_OPTIONS: '--require=unknown', BUN_OPTIONS: '--preload=unknown', ANTHROPIC_BASE_URL: 'https://custom-provider.example.test' })) { test(`${key}=${value} is fresh or ineligible`, () => { - const f = fixture(); f.cache().publish(scores); + const f = fixture(); f.cache().publish(panel); f.opts.env = { ...f.env, [key]: value }; const cache = f.cache(); - expect(cache.lookup()).toBeNull(); cache.publish(scores); expect(f.entries()).toHaveLength(1); + expect(cache.lookup()).toBeNull(); cache.publish(panel); expect(f.entries()).toHaveLength(1); }); } test('runtime/model/threshold changes miss, and retries never reuse or publish', () => { - const f = fixture(); f.cache().publish(scores); + const f = fixture(); f.cache().publish(panel); for (const overrides of [{ GSTACK_EVAL_MODEL_JUDGE: 'different-model' }, { EVALS_CACHE_RUNTIME_ID: 'c'.repeat(64) }]) { f.opts.env = { ...f.env, ...overrides }; expect(f.cache().lookup()).toBeNull(); } f.opts.env = f.env; f.opts.thresholds.clarity = 5; expect(f.cache().lookup()).toBeNull(); f.opts.thresholds.clarity = 4; f.opts.attempt = 2; const retry = f.cache(); - expect(retry.lookup()).toBeNull(); retry.publish(scores); expect(f.entries()).toHaveLength(1); + expect(retry.lookup()).toBeNull(); retry.publish(panel); expect(f.entries()).toHaveLength(1); }); test('frontier reader calibration cannot reuse a score from the unspecified-reader rubric', () => { - const f = fixture(); f.cache().publish(scores); + const f = fixture(); f.cache().publish(panel); const original = f.opts.prompt; f.opts.agentCapability = 'frontier'; f.refreshPrompt(); expect(f.opts.prompt).not.toBe(original); expect(f.cache().lookup()).toBeNull(); - f.cache().publish(scores); + f.cache().publish(panel); expect(f.entries()).toHaveLength(2); - expect(f.cache().lookup()?.scores).toEqual(scores); + expect(f.cache().lookup()?.samples).toEqual(panel); delete f.opts.agentCapability; f.refreshPrompt(); expect(f.opts.prompt).toBe(original); - expect(f.cache().lookup()?.scores).toEqual(scores); + expect(f.cache().lookup()?.samples).toEqual(panel); }); test('a pinned workflow judge model overrides the global model and changes the cache identity', () => { const f = fixture(); f.opts.model = 'claude-sonnet-4-6'; - f.cache().publish(scores); + f.cache().publish(panel); expect(f.entries()).toHaveLength(1); f.opts.env = { ...f.env, GSTACK_EVAL_MODEL_JUDGE: 'different-global-model' }; - expect(f.cache().lookup()?.scores).toEqual(scores); + expect(f.cache().lookup()?.samples).toEqual(panel); f.opts.model = 'claude-opus-4-7'; expect(f.cache().lookup()).toBeNull(); }); test('failed assertions, missing provenance, and missing imported dependencies cannot supply a receipt', () => { - const f = fixture(); f.cache().publish({ ...scores, clarity: 3 }); expect(f.entries()).toHaveLength(0); - f.opts.env = { ...f.env, EVALS_RUN_ID: '' }; f.cache().publish(scores); expect(f.entries()).toHaveLength(0); + const f = fixture(); f.cache().publish(panelOf({ ...scores, clarity: 3 })); expect(f.entries()).toHaveLength(0); + f.opts.env = { ...f.env, EVALS_RUN_ID: '' }; f.cache().publish(panel); expect(f.entries()).toHaveLength(0); f.opts.env = f.env; fs.unlinkSync(path.join(f.root, 'test/helpers/nested.ts')); - f.cache().publish(scores); expect(f.entries()).toHaveLength(0); + f.cache().publish(panel); expect(f.entries()).toHaveLength(0); }); test('cached payload schema remains small and cannot carry operational fields', () => { @@ -167,8 +171,9 @@ test('workflow registration preserves model work and reserves only terminal-reco const source = fs.readFileSync(path.join(import.meta.dir, 'skill-llm-eval.test.ts'), 'utf8'); const body = source.split('async function runWorkflowJudge')[1]!.split('// Block 1:')[0]!; const stages = ['workflowJudgeAttempts.set', 'readWorkflowJudgeInput(', 'cache.lookup()', - 'callJudge(prompt, opts.model, { signal: controller.signal, max_tokens: maxTokens,', - 'expect(scores.clarity)', 'expect(scores.completeness)', 'expect(scores.actionability)', 'cache.publish(scores, active)'] + 'judgePanel(() => callJudge(prompt, opts.model, { signal: controller.signal, max_tokens: maxTokens,', + 'scores = judgePanelMean(samples, JUDGE_SCORE_DIMENSIONS);', + 'expect(scores.clarity)', 'expect(scores.completeness)', 'expect(scores.actionability)', 'cache.publish(samples, active)'] .map(stage => body.indexOf(stage)); expect(stages.every(position => position >= 0)).toBe(true); expect(stages).toEqual([...stages].sort((a, b) => a - b)); @@ -203,6 +208,7 @@ function actualCallback(f: ReturnType, overrides: { 'evalCollector', 'expect', 'console', 'performance', 'JUDGE_MS', 'WORKFLOW_JUDGE_RECORD_MS', 'setTimeout', 'clearTimeout', 'JudgeRefusalError', 'getCookieWorkflowManualReview', 'DEFAULT_JUDGE_MAX_TOKENS', 'resolveEvalModel', 'WORKFLOW_JUDGE_RESPONSE_SCHEMA', 'validWorkflowJudgeScore', + 'judgePanel', 'judgePanelMean', 'judgePanelReasoning', 'JUDGE_SCORE_DIMENSIONS', `${javascript}\nreturn runWorkflowJudge;`)( f.root, overrides.read ?? readWorkflowJudgeInput, buildWorkflowJudgePrompt, (options: WorkflowCacheOptions) => (overrides.prepare ?? prepareWorkflowJudgeCache)({ ...options, env: f.env }), @@ -213,7 +219,8 @@ function actualCallback(f: ReturnType, overrides: { overrides.clock ? { now: overrides.clock } : performance, overrides.budget ?? 120_000, overrides.allowance ?? 5_000, overrides.setTimer ?? setTimeout, overrides.clearTimer ?? clearTimeout, JudgeRefusalError, getCookieWorkflowManualReview, DEFAULT_JUDGE_MAX_TOKENS, resolveEvalModel, - WORKFLOW_JUDGE_RESPONSE_SCHEMA, validWorkflowJudgeScore); + WORKFLOW_JUDGE_RESPONSE_SCHEMA, validWorkflowJudgeScore, + judgePanel, judgePanelMean, judgePanelReasoning, JUDGE_SCORE_DIMENSIONS); return { run, records, signals, prompts, attempts, options: { ...f.opts, suite: 'Cache regression' } }; } @@ -223,7 +230,7 @@ test('the actual workflow callback preserves the pinned model and frontier rubri const actual = actualCallback(f, { judge: async (_prompt, model) => { models.push(model); return scores; } }); await actual.run({ ...actual.options, model: 'claude-sonnet-4-6', agentCapability: 'frontier', readInput: () => readWorkflowJudgeInput(f.opts) }); - expect(models).toEqual(['claude-sonnet-4-6']); + expect(models).toEqual(Array(SAMPLES).fill('claude-sonnet-4-6')); expect(actual.prompts[0]).toContain('GPT-5.6 Sol-level capability or stronger'); expect(actual.records[0]).toMatchObject({ passed: true, model: 'claude-sonnet-4-6', prompt: actual.prompts[0] }); }); @@ -239,7 +246,7 @@ test.each(['ship', 'review'])('the registered %s callback sends the frontier rub endMarker: f.opts.endMarker, references: [] }; const passing = actualCallback(f, { judge: async () => ({ ...scores, clarity: 3 }) }); await passing.run(options); - expect(passing.prompts).toHaveLength(1); + expect(passing.prompts).toHaveLength(SAMPLES); expect(passing.prompts[0]).toContain('GPT-5.6 Sol-level capability or stronger'); expect(passing.records[0]).toMatchObject({ passed: true, execution: 'executed', judge_scores: { clarity: 3 } }); const failing = actualCallback(f, { judge: async () => ({ ...scores, clarity: 2 }) }); @@ -254,8 +261,8 @@ test('the actual workflow callback executes once, reuses with provenance, and pr const f = fixture(); const first = actualCallback(f); const options = { ...f.opts, suite: 'Cache regression' }; await first.run(options); - expect(first.prompts).toEqual([f.opts.prompt]); expect(f.entries()).toHaveLength(1); - expect(first.records[0]).toMatchObject({ passed: true, execution: 'executed', cost_usd: 0.02 }); + expect(first.prompts).toEqual(Array(SAMPLES).fill(f.opts.prompt)); expect(f.entries()).toHaveLength(1); + expect(first.records[0]).toMatchObject({ passed: true, execution: 'executed', cost_usd: 0.02 * SAMPLES }); expect(first.records[0]).not.toHaveProperty('prompt'); expect(first.records[0]).not.toHaveProperty('model'); const reused = actualCallback(f, { judge: async () => ({ ...scores, clarity: 1 }) }); @@ -312,13 +319,13 @@ test('a superseding attempt cancels its predecessor before either can record a s }); test('a failed input read consumes attempt one and prevents a retry from borrowing or publishing a receipt', async () => { - const f = fixture(); f.cache().publish(scores); const receipt = fs.readFileSync(path.join(f.env.EVALS_CACHE_DIR, f.entries()[0]), 'utf8'); + const f = fixture(); f.cache().publish(panel); const receipt = fs.readFileSync(path.join(f.env.EVALS_CACHE_DIR, f.entries()[0]), 'utf8'); let reads = 0; const h = actualCallback(f, { read: options => { if (++reads === 1) throw new Error('Missing workflow fixture'); return readWorkflowJudgeInput(options); } }); await expect(h.run(h.options)).rejects.toThrow('Missing workflow fixture'); expect(h.records[0]).toMatchObject({ passed: false, exit_reason: 'harness_error' }); await h.run(h.options); - expect(h.prompts).toHaveLength(1); + expect(h.prompts).toHaveLength(SAMPLES); expect(h.records.map(record => record.execution)).toEqual(['executed', 'executed']); expect(h.attempts.get(f.opts.testName).attempt).toBe(2); expect(fs.readFileSync(path.join(f.env.EVALS_CACHE_DIR, f.entries()[0]), 'utf8')).toBe(receipt); @@ -333,14 +340,14 @@ test('monotonic expiry after a synchronous preparation or late model response re await expect(h.run(h.options)).rejects.toThrow('deadline'); expect(h.records).toHaveLength(1); expect(h.records[0]).toMatchObject({ passed: false, exit_reason: 'timeout', duration_ms: 21 }); - expect(h.prompts).toHaveLength(phase === 'preparation' ? 0 : 1); + expect(h.prompts).toHaveLength(phase === 'preparation' ? 0 : SAMPLES); expect(f.entries()).toHaveLength(0); } }); test('publication rechecks after input scanning and withdraws a receipt if recording expires', async () => { const f = fixture(); let checks = 0; - f.cache().publish(scores, () => ++checks < 2); + f.cache().publish(panel, () => ++checks < 2); expect(checks).toBe(2); expect(f.entries()).toHaveLength(0); let now = 0; const h = actualCallback(f, { budget: 20, allowance: 5, clock: () => now, @@ -373,7 +380,7 @@ test('the actual workflow callback preserves the complete public API body; cance try { const h = actualCallback(f, { judge: (prompt, model, options) => callJudge(prompt, model, options) }); await h.run(h.options); - expect(create).toHaveBeenCalledTimes(1); + expect(create).toHaveBeenCalledTimes(SAMPLES); expect(create.mock.calls[0]).toEqual([{ model: resolveEvalModel('judge'), max_tokens: 8192, messages: [{ role: 'user', content: f.opts.prompt }], @@ -412,40 +419,40 @@ test('Ship sends its authorized 64k cap and compact response contract through th f.opts.structuredResponse = true; f.opts.maxTokens = 65_536; f.opts.stream = true; - expect(f.cache().lookup()?.scores).toEqual(scores); + expect(f.cache().lookup()?.samples).toEqual(panel); } finally { stream.mockRestore(); } }); test('changing response serialization misses the cache even when prompt and model match', () => { - const f = fixture(); f.cache().publish(scores); + const f = fixture(); f.cache().publish(panel); f.opts.structuredResponse = true; expect(f.cache().lookup()).toBeNull(); - f.cache().publish(scores); + f.cache().publish(panel); expect(f.entries()).toHaveLength(2); - expect(f.cache().lookup()?.scores).toEqual(scores); + expect(f.cache().lookup()?.samples).toEqual(panel); const description = WORKFLOW_JUDGE_RESPONSE_SCHEMA.properties.reasoning.description; try { WORKFLOW_JUDGE_RESPONSE_SCHEMA.properties.reasoning.description += ' Changed response contract.'; expect(f.cache().lookup()).toBeNull(); } finally { WORKFLOW_JUDGE_RESPONSE_SCHEMA.properties.reasoning.description = description; } - expect(f.cache().lookup()?.scores).toEqual(scores); + expect(f.cache().lookup()?.samples).toEqual(panel); f.opts.structuredResponse = false; - expect(f.cache().lookup()?.scores).toEqual(scores); + expect(f.cache().lookup()?.samples).toEqual(panel); }); test('the actual cap and streaming transport independently affect workflow cache identity', () => { - const f = fixture(); f.cache().publish(scores); + const f = fixture(); f.cache().publish(panel); f.opts.maxTokens = 65_536; expect(f.cache().lookup()).toBeNull(); - f.cache().publish(scores); + f.cache().publish(panel); f.opts.stream = true; expect(f.cache().lookup()).toBeNull(); - f.cache().publish(scores); + f.cache().publish(panel); expect(f.entries()).toHaveLength(3); - expect(f.cache().lookup()?.scores).toEqual(scores); + expect(f.cache().lookup()?.samples).toEqual(panel); delete f.opts.maxTokens; delete f.opts.stream; - expect(f.cache().lookup()?.scores).toEqual(scores); + expect(f.cache().lookup()?.samples).toEqual(panel); }); test('the structured callback rejects incomplete, schema-invalid and below-threshold answers without cache credit', async () => { @@ -473,3 +480,96 @@ test('the structured callback rejects incomplete, schema-invalid and below-thres expect(validWorkflowJudgeScore({ ...scores, reasoning: Array(149).fill('word').join(' ') }, { clarity: 1, completeness: 1, actionability: 1 }, true)).toBe(true); } finally { stream.mockRestore(); diagnostics.mockRestore(); } }); + +// --- Judge panel policy (EVAL_POLICY.judge): fixed concurrent samples, per-dimension +// mean and boolean majority against unchanged thresholds, an erroring sample fails +// the whole panel and is never resampled. The provider is always a stub. +const panelScore = (clarity: number, completeness = 4, actionability = 4) => ({ clarity, completeness, actionability, reasoning: `c${clarity}` }); +const panelThresholds = { clarity: 3, completeness: 3, actionability: 4 }; +const panelRefusal = () => new JudgeRefusalError({ id: 'msg_1', _request_id: 'req_1', model: 'm', usage: { input_tokens: 1, output_tokens: 0 }, content: [] }); + +describe('judge panel', () => { + test('the pre-registered panel is three samples, and the helper restates EVAL_POLICY exactly', () => { + expect(EVAL_POLICY.judge.samples).toBe(3); + expect(JUDGE_PANEL_SAMPLES).toBe(EVAL_POLICY.judge.samples); + }); + + test('draws every sample concurrently before any resolves', async () => { + let started = 0; + const releases: Array<() => void> = []; + const panel = judgePanel(() => new Promise(resolve => { started += 1; releases.push(() => resolve(started)); })); + await Promise.resolve(); + expect(started).toBe(SAMPLES); + releases.forEach(release => release()); + expect(await panel).toHaveLength(SAMPLES); + }); + + test('an erroring sample fails the panel and is never resampled', async () => { + let calls = 0; + const panel = judgePanel(async () => { + calls += 1; + if (calls === 2) throw new Error('Judge returned non-JSON: nope'); + return panelScore(5); + }); + await expect(panel).rejects.toThrow('non-JSON'); + expect(calls).toBe(SAMPLES); + }); + + test('a refusal on every sample stays a provider refusal; a partial refusal is an ordinary failure', async () => { + await expect(judgePanel(async () => { throw panelRefusal(); })).rejects.toBeInstanceOf(JudgeRefusalError); + let calls = 0; + const partial = judgePanel(async () => { if (++calls === 1) throw panelRefusal(); return panelScore(4); }); + const error = await partial.then(() => null, (reason: unknown) => reason); + expect(error).toBeInstanceOf(Error); + expect(error).not.toBeInstanceOf(JudgeRefusalError); + expect(String(error)).toContain(`sample 1 of ${SAMPLES} failed beside scored samples`); + }); + + test('numeric dimensions gate on the per-dimension mean; one low sample can be outvoted, a low mean cannot', () => { + const outvoted = judgePanelMean([panelScore(2), panelScore(4), panelScore(4)], JUDGE_SCORE_DIMENSIONS); + expect(outvoted.clarity).toBeCloseTo(10 / 3); + expect(outvoted.clarity).toBeGreaterThanOrEqual(panelThresholds.clarity); + const low = judgePanelMean([panelScore(2), panelScore(2), panelScore(4)], JUDGE_SCORE_DIMENSIONS); + expect(low.clarity).toBeLessThan(panelThresholds.clarity); + // No compensation across dimensions: each is averaged on its own. + expect(judgePanelMean([panelScore(5, 1), panelScore(5, 1), panelScore(5, 1)], JUDGE_SCORE_DIMENSIONS).completeness).toBe(1); + }); + + test('malformed sample fields fail the panel instead of averaging to NaN', () => { + expect(() => judgePanelMean([panelScore(4), { ...panelScore(4), clarity: '4' as unknown as number }, panelScore(4)], JUDGE_SCORE_DIMENSIONS)).toThrow('sample 2 has non-numeric clarity'); + expect(() => judgePanelMean([panelScore(4), null as unknown as ReturnType], JUDGE_SCORE_DIMENSIONS)).toThrow('sample 2'); + expect(() => judgePanelMean([], JUDGE_SCORE_DIMENSIONS)).toThrow('no samples'); + }); + + test('boolean fields gate on a strict majority', () => { + const vote = (...values: boolean[]) => judgePanelMajority(values.map(value => ({ ok: value })), 'ok'); + expect(vote(true, true, false)).toBe(true); + expect(vote(true, false, false)).toBe(false); + expect(vote(true, false)).toBe(false); + expect(() => judgePanelMajority([{ ok: true }, { ok: 'yes' }], 'ok')).toThrow('sample 2 has non-boolean ok'); + }); + + test('reasoning keeps every sample, numbered, even for malformed samples', () => { + expect(judgePanelReasoning([panelScore(4), null, { reasoning: 7 }])).toBe('[sample 1] c4\n[sample 2] \n[sample 3] '); + }); + + test('the cache stores and validates only a complete panel against the mean', () => { + expect(validWorkflowJudgePanel({ samples: [panelScore(2), panelScore(4), panelScore(4)] }, panelThresholds)).toBe(true); + expect(validWorkflowJudgePanel({ samples: [panelScore(2), panelScore(2), panelScore(4)] }, panelThresholds)).toBe(false); + expect(validWorkflowJudgePanel({ samples: [panelScore(4), panelScore(4)] }, panelThresholds)).toBe(false); + expect(validWorkflowJudgePanel({ samples: [panelScore(4), panelScore(4), panelScore(4), panelScore(4)] }, panelThresholds)).toBe(false); + expect(validWorkflowJudgePanel({ samples: [panelScore(4), panelScore(4), { ...panelScore(4), clarity: 6 }] }, panelThresholds)).toBe(false); + expect(validWorkflowJudgePanel({ samples: [panelScore(4), panelScore(4), panelScore(4)], prompt: 'x' }, panelThresholds)).toBe(false); + expect(validWorkflowJudgePanel(panelScore(4), panelThresholds)).toBe(false); + }); + + test('every judge in the quality file samples through the panel, never a lone call', () => { + const source = fs.readFileSync(path.join(import.meta.dir, 'skill-llm-eval.test.ts'), 'utf8'); + const calls = [...source.matchAll(/\b(?:callJudge<[^>(]*(?:<[^>]*>[^>(]*)*>|judge)\(/g)]; + expect(calls.length).toBeGreaterThanOrEqual(8); + for (const call of calls) { + expect(source.slice(Math.max(0, call.index! - 25), call.index), `unpaneled judge call at offset ${call.index}`).toMatch(/judgePanel\(\(\) => $/); + } + expect(source).not.toMatch(/\bscores\.reasoning\b|\bresult\.reasoning\b/); + }); +}); From 0292ee3f6f8a14f51d6a73abf4d163b19a91b528 Mon Sep 17 00:00:00 2001 From: garrytan Date: Tue, 29 Sep 2026 19:06:42 +0000 Subject: [PATCH 06/15] test(evals): classify every live case and re-select a case when its kind changes E2E_KINDS: rule by default (191 E2E ids), 22 behavior cases whose verdict is a live model choice with an acceptable sub-100% per-trial rate, each with a BEHAVIOR_WHY tolerance, and 25 judge entries (the 24 workflow judges plus the fixed-fixture llm-judge-recommendation rubric check). Contract-shaped cases (ask-before-decide, plan-mode no-writes, mandated steps, secrets, the batching floor) stay rule. Behavior requires a known literal registration and an exact Bun test name so the case runs as its own trial shard. Map-diff selection now diffs E2E_KINDS and BEHAVIOR_WHY per key, and a base revision without them selects every key, so a kind flip runs the panel it introduces. test/eval-kinds.test.ts enforces coverage, tolerances, isolatability and the reviewed counts, printing the literal to add. --- test/eval-kinds.test.ts | 107 ++++++++++++++++++++++++ test/helpers/test-selection.ts | 19 ++++- test/helpers/touchfiles-data.ts | 141 +++++++++++++++++++++----------- 3 files changed, 216 insertions(+), 51 deletions(-) create mode 100644 test/eval-kinds.test.ts diff --git a/test/eval-kinds.test.ts b/test/eval-kinds.test.ts new file mode 100644 index 000000000..ea6480f6c --- /dev/null +++ b/test/eval-kinds.test.ts @@ -0,0 +1,107 @@ +/** + * Eval kind registry (E2E_KINDS / BEHAVIOR_WHY in touchfiles-data.ts). The + * kind fixes a case's trial policy before the run, so the registry must cover + * every live case exactly once, every behavior case must name its tolerated + * deviation, and a behavior case must be isolatable as its own trial shard. + * A kind edit must re-select the case in the PR lane (map-diff). + */ +import { describe, expect, test } from 'bun:test'; +import * as fs from 'node:fs'; +import * as path from 'node:path'; + +import { BEHAVIOR_WHY, E2E_KINDS, E2E_TIERS, E2E_TOUCHFILES, LLM_JUDGE_TOUCHFILES } from './helpers/touchfiles-data'; +import { diffTouchfileMapsCore, type TouchfileMaps } from './helpers/test-selection'; +import { CASE_TEST_NAMES, fileCaseRegistration } from '../scripts/test-paid-shards'; +import { isPaidTestFile } from './helpers/paid-test-set'; + +const ROOT = path.resolve(import.meta.dir, '..'); +const KIND_RULE = "Pick the kind by what can make the verdict differ between two runs of the same commit: 'rule' when nothing " + + "stochastic decides it or it checks a contract the product must meet every run (the default); 'behavior' when a live " + + "model choice decides it and a sub-100% per-trial rate is acceptable (add a BEHAVIOR_WHY line); 'judge' when the only " + + 'stochastic step is an LLM judge scoring a fixed input.'; + +const liveIds = [...Object.keys(E2E_TIERS), ...Object.keys(LLM_JUDGE_TOUCHFILES)]; +const behaviorIds = Object.keys(E2E_KINDS).filter(id => E2E_KINDS[id] === 'behavior').sort(); + +describe('E2E_KINDS registry', () => { + test('every live case has exactly one kind and no kind names a dead case', () => { + const missing = liveIds.filter(id => !(id in E2E_KINDS)); + expect(missing.length, missing.length ? `add to E2E_KINDS:\n${missing.map(id => ` '${id}': 'rule', // `).join('\n')}\n${KIND_RULE}` : '').toBe(0); + const unknown = Object.keys(E2E_KINDS).filter(id => !liveIds.includes(id)); + expect(unknown, `E2E_KINDS names ids that are neither E2E_TIERS nor LLM_JUDGE_TOUCHFILES keys`).toEqual([]); + expect(new Set(liveIds).size).toBe(liveIds.length); + }); + + test('kinds are rule, behavior or judge; every LLM-judge entry is judge-kind', () => { + for (const [id, kind] of Object.entries(E2E_KINDS)) expect(['rule', 'behavior', 'judge'], id).toContain(kind); + for (const id of Object.keys(LLM_JUDGE_TOUCHFILES)) expect(E2E_KINDS[id], `${id}: a workflow judge scores a fixed input`).toBe('judge'); + }); + + test('BEHAVIOR_WHY names the tolerance of exactly the behavior cases', () => { + expect(Object.keys(BEHAVIOR_WHY).sort()).toEqual(behaviorIds); + for (const id of behaviorIds) { + expect(BEHAVIOR_WHY[id]!.trim().length, `${id}: BEHAVIOR_WHY must say why an occasional deviation is acceptable`).toBeGreaterThanOrEqual(30); + } + }); + + test('a behavior case is an isolatable trial shard: known literal registration and an exact Bun test name', () => { + for (const id of behaviorIds) { + const files = E2E_TOUCHFILES[id]!.filter(file => /^test\/[^/]+\.test\.ts$/.test(file) && isPaidTestFile(file)); + expect(files.length, `${id}: no paid test file registers it`).toBeGreaterThan(0); + for (const file of files) { + const source = fs.readFileSync(path.join(ROOT, file), 'utf8'); + expect(fileCaseRegistration(file, source).known, `${id}: ${file} has a computed registration; behavior needs a literal one`).toBe(true); + const name = CASE_TEST_NAMES[id] ?? id; + const literal = new RegExp(`\\b(?:test(?:\\.serial|\\.concurrent)?|testIfSelected|testConcurrentIfSelected)\\(\\s*(['"\`])${name.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}\\1`); + expect(literal.test(source), `${id}: ${file} must register the Bun test named '${name}'`).toBe(true); + } + } + }); + + test('the classification is the reviewed one: rule by default, 22 behavior, 25 judge', () => { + const counts = Object.values(E2E_KINDS).reduce>((acc, kind) => ({ ...acc, [kind]: (acc[kind] ?? 0) + 1 }), {}); + expect(counts).toEqual({ rule: liveIds.length - 22 - 25, behavior: 22, judge: 25 }); + // Contract-shaped cases stay rule: ask-before-decide, plan-mode no-writes, + // mandated steps, secrets, and the batching floor never ride a majority. + for (const id of ['plan-ceo-mode-routing', 'plan-eng-multi-finding-batching', 'plan-design-review-plan-mode', + 'plan-eng-review-plan-mode', 'plan-ceo-section-loading', 'setup-gbrain-bad-token', 'qa-only-no-fix', 'review-sql-injection']) { + expect(E2E_KINDS[id], id).toBe('rule'); + } + }); +}); + +describe('kind edits re-select their case (map-diff)', () => { + const base = (): TouchfileMaps => ({ + E2E_TOUCHFILES: { alpha: ['a/**'], beta: ['b/**'] }, + E2E_TIERS: { alpha: 'gate', beta: 'periodic' }, + LLM_JUDGE_TOUCHFILES: { 'judge one': ['j/SKILL.md'] }, + GLOBAL_TOUCHFILES: [], + E2E_KINDS: { alpha: 'rule', beta: 'rule', 'judge one': 'judge' }, + BEHAVIOR_WHY: {}, + }); + + test('a rule -> behavior flip selects exactly that case', () => { + const next = base(); + next.E2E_KINDS = { ...next.E2E_KINDS, beta: 'behavior' }; + next.BEHAVIOR_WHY = { beta: 'tolerated deviation' }; + expect(diffTouchfileMapsCore(base(), next).changedTests).toEqual(['beta']); + }); + + test('a BEHAVIOR_WHY edit alone selects its case', () => { + const old = base(); old.E2E_KINDS!.beta = 'behavior'; old.BEHAVIOR_WHY = { beta: 'one' }; + const next = base(); next.E2E_KINDS!.beta = 'behavior'; next.BEHAVIOR_WHY = { beta: 'two' }; + expect(diffTouchfileMapsCore(old, next).changedTests).toEqual(['beta']); + }); + + test('a base revision without the kind maps selects every key', () => { + const old = base(); delete old.E2E_KINDS; delete old.BEHAVIOR_WHY; + expect(diffTouchfileMapsCore(old, base()).changedTests).toEqual(['alpha', 'beta', 'judge one']); + }); + + test('dropping a kind entry while the case lives on counts as changed, not removed', () => { + const next = base(); delete next.E2E_KINDS!.alpha; + const result = diffTouchfileMapsCore(base(), next); + expect(result.changedTests).toEqual(['alpha']); + expect(result.removedTests).toEqual([]); + }); +}); diff --git a/test/helpers/test-selection.ts b/test/helpers/test-selection.ts index 13d04cb6b..51655f89f 100644 --- a/test/helpers/test-selection.ts +++ b/test/helpers/test-selection.ts @@ -34,6 +34,8 @@ import { E2E_TIERS, LLM_JUDGE_TOUCHFILES, GLOBAL_TOUCHFILES, + E2E_KINDS, + BEHAVIOR_WHY, } from './touchfiles-data'; /** Repo-relative path of the pure-data file (the map-diff subject). */ @@ -145,6 +147,9 @@ export interface TouchfileMaps { E2E_TIERS: Record; LLM_JUDGE_TOUCHFILES: Record; GLOBAL_TOUCHFILES: string[]; + /** Absent on base revisions older than the eval-kind registry: every current key then counts as changed. */ + E2E_KINDS?: Record; + BEHAVIOR_WHY?: Record; } export type MapDiffCause = @@ -171,6 +176,8 @@ const CURRENT_MAPS: TouchfileMaps = { E2E_TIERS, LLM_JUDGE_TOUCHFILES, GLOBAL_TOUCHFILES, + E2E_KINDS, + BEHAVIOR_WHY, }; function isStringArray(v: unknown): v is string[] { @@ -193,14 +200,18 @@ function isTouchfileMaps(v: unknown): v is TouchfileMaps { return isRecordOfStringArrays(o.E2E_TOUCHFILES) && isRecordOfStrings(o.E2E_TIERS) && isRecordOfStringArrays(o.LLM_JUDGE_TOUCHFILES) - && isStringArray(o.GLOBAL_TOUCHFILES); + && isStringArray(o.GLOBAL_TOUCHFILES) + && (o.E2E_KINDS === undefined || isRecordOfStrings(o.E2E_KINDS)) + && (o.BEHAVIOR_WHY === undefined || isRecordOfStrings(o.BEHAVIOR_WHY)); } /** * Pure map-diff core (injectable for tests — no git, no filesystem). * * A key counts as CHANGED when it was added to any per-key map, its dep-list - * array differs, or its tier value flipped. A key counts as REMOVED only when + * array differs, or its tier, kind or behavior tolerance changed. A per-key + * map missing on the old side (a base revision older than E2E_KINDS / + * BEHAVIOR_WHY) makes every key of that map count as added. A key counts as REMOVED only when * it is gone from every new per-key map; a key dropped from one map but still * present in another (e.g. tier entry deleted, touchfile entry kept) counts * as changed — conservative, because the test still exists with a different @@ -212,7 +223,7 @@ export function diffTouchfileMapsCore( oldMaps: TouchfileMaps, newMaps: TouchfileMaps, ): { changedTests: string[]; removedTests: string[]; globalTouchfilesChanged: boolean } { - const perKeyMapNames = ['E2E_TOUCHFILES', 'E2E_TIERS', 'LLM_JUDGE_TOUCHFILES'] as const; + const perKeyMapNames = ['E2E_TOUCHFILES', 'E2E_TIERS', 'LLM_JUDGE_TOUCHFILES', 'E2E_KINDS', 'BEHAVIOR_WHY'] as const; const changed = new Set(); const rawRemoved = new Set(); @@ -290,6 +301,8 @@ export function diffTouchfileMaps( ' E2E_TIERS: m.E2E_TIERS,', ' LLM_JUDGE_TOUCHFILES: m.LLM_JUDGE_TOUCHFILES,', ' GLOBAL_TOUCHFILES: m.GLOBAL_TOUCHFILES,', + ' E2E_KINDS: m.E2E_KINDS,', + ' BEHAVIOR_WHY: m.BEHAVIOR_WHY,', '}));', '', ].join('\n')); diff --git a/test/helpers/touchfiles-data.ts b/test/helpers/touchfiles-data.ts index 18fc17f77..860afc9f2 100644 --- a/test/helpers/touchfiles-data.ts +++ b/test/helpers/touchfiles-data.ts @@ -1597,7 +1597,7 @@ export const E2E_KINDS: Record = { 'shared-libs-unsupported-git': 'rule', 'shared-libs-review-lifecycle': 'rule', 'shared-libs-review-revalidation': 'rule', - 'shared-libs-opportunity-judgment': 'rule', + 'shared-libs-opportunity-judgment': 'behavior', 'shared-libs-pr-coverage': 'rule', 'shared-libs-plan-callers': 'rule', 'browse-basic': 'rule', @@ -1634,7 +1634,7 @@ export const E2E_KINDS: Record = { 'review-sql-injection': 'rule', 'review-enum-completeness': 'rule', 'review-base-branch': 'rule', - 'review-design-lite': 'rule', + 'review-design-lite': 'behavior', 'review-coverage-audit': 'rule', 'review-dashboard-via': 'rule', 'review-army-migration-safety': 'rule', @@ -1642,21 +1642,21 @@ export const E2E_KINDS: Record = { 'review-army-delivery-audit': 'rule', 'review-army-quality-score': 'rule', 'review-army-json-findings': 'rule', - 'review-army-red-team': 'rule', - 'review-army-consensus': 'rule', - 'review-army-simplification': 'rule', - 'review-army-simplification-precision': 'rule', + 'review-army-red-team': 'behavior', + 'review-army-consensus': 'behavior', + 'review-army-simplification': 'behavior', + 'review-army-simplification-precision': 'behavior', 'office-hours-spec-review': 'rule', - 'office-hours-brain-writeback': 'rule', + 'office-hours-brain-writeback': 'behavior', 'gbrain-roundtrip-local': 'rule', 'sync-gbrain-read-ready': 'rule', 'sync-gbrain-read-unknown': 'rule', - 'office-hours-forcing-energy': 'rule', - 'office-hours-builder-wildness': 'rule', + 'office-hours-forcing-energy': 'behavior', + 'office-hours-builder-wildness': 'behavior', 'plan-ceo-review': 'rule', 'plan-ceo-review-selective': 'rule', 'plan-ceo-review-benefits': 'rule', - 'plan-ceo-review-expansion-energy': 'rule', + 'plan-ceo-review-expansion-energy': 'behavior', 'plan-eng-review': 'rule', 'plan-eng-review-artifact': 'rule', 'plan-eng-coverage-audit': 'rule', @@ -1688,16 +1688,16 @@ export const E2E_KINDS: Record = { 'setup-gbrain-remote': 'rule', 'setup-gbrain-bad-token': 'rule', 'setup-gbrain-path4-local-pglite': 'rule', - 'plan-ceo-review-format-mode': 'rule', - 'plan-ceo-review-format-approach': 'rule', - 'plan-eng-review-format-coverage': 'rule', - 'plan-eng-review-format-kind': 'rule', - 'office-hours-phase4-fork': 'rule', - 'llm-judge-recommendation': 'rule', - 'plan-ceo-review-prosons-cadence': 'rule', - 'plan-review-prosons-format': 'rule', - 'plan-review-prosons-hardstop-neg': 'rule', - 'plan-review-prosons-neutral-neg': 'rule', + 'plan-ceo-review-format-mode': 'behavior', + 'plan-ceo-review-format-approach': 'behavior', + 'plan-eng-review-format-coverage': 'behavior', + 'plan-eng-review-format-kind': 'behavior', + 'office-hours-phase4-fork': 'behavior', + 'llm-judge-recommendation': 'judge', + 'plan-ceo-review-prosons-cadence': 'behavior', + 'plan-review-prosons-format': 'behavior', + 'plan-review-prosons-hardstop-neg': 'behavior', + 'plan-review-prosons-neutral-neg': 'behavior', 'plan-tune-inspect': 'rule', 'codex-offered-office-hours': 'rule', 'codex-offered-ceo-review': 'rule', @@ -1758,7 +1758,7 @@ export const E2E_KINDS: Record = { 'design-review-detector-shim': 'rule', 'design-review-detector-shim-dom': 'rule', 'design-review-plugin-handoff': 'rule', - 'design-html-slop-gate': 'rule', + 'design-html-slop-gate': 'behavior', 'diagram-triplet': 'rule', 'diagram-authoring-quality': 'rule', 'gstack-upgrade-happy-path': 'rule', @@ -1770,8 +1770,8 @@ export const E2E_KINDS: Record = { 'setup-deploy-workflow': 'rule', 'autoplan-dual-voice': 'rule', 'benchmark-providers-live': 'rule', - 'scrape-match-path': 'rule', - 'scrape-prototype-path': 'rule', + 'scrape-match-path': 'behavior', + 'scrape-prototype-path': 'behavior', 'skillify-happy-path': 'rule', 'skillify-provenance-refusal': 'rule', 'skillify-approval-reject': 'rule', @@ -1799,30 +1799,30 @@ export const E2E_KINDS: Record = { 'overlay-harness-opus-4-7-literal-interpretation': 'rule', 'overlay-harness-claude-dedicated-tools-vs-bash-sonnet': 'rule', 'journey-negatives': 'rule', - 'review/SKILL.md workflow': 'rule', - 'setup-browser-cookies/SKILL.md workflow': 'rule', - 'browse/SKILL.md reference': 'rule', - 'setup block': 'rule', - 'qa/SKILL.md workflow': 'rule', - 'qa/SKILL.md health rubric': 'rule', - 'qa/SKILL.md anti-refusal': 'rule', - 'cross-skill greptile consistency': 'rule', - 'ship/SKILL.md workflow': 'rule', - 'document-release/SKILL.md workflow': 'rule', - 'plan-ceo-review/SKILL.md modes': 'rule', - 'plan-eng-review/SKILL.md sections': 'rule', - 'plan-design-review/SKILL.md passes': 'rule', - 'design-review/SKILL.md fix loop': 'rule', - 'design-consultation/SKILL.md research': 'rule', - 'land-and-deploy/SKILL.md workflow': 'rule', - 'canary/SKILL.md monitoring loop': 'rule', - 'benchmark/SKILL.md perf collection': 'rule', - 'setup-deploy/SKILL.md platform setup': 'rule', - 'retro/SKILL.md instructions': 'rule', - 'qa-only/SKILL.md workflow': 'rule', - 'gstack-upgrade/SKILL.md upgrade flow': 'rule', - 'sync-gbrain/SKILL.md read-only readiness': 'rule', - 'voice directive tone': 'rule', + 'review/SKILL.md workflow': 'judge', + 'setup-browser-cookies/SKILL.md workflow': 'judge', + 'browse/SKILL.md reference': 'judge', + 'setup block': 'judge', + 'qa/SKILL.md workflow': 'judge', + 'qa/SKILL.md health rubric': 'judge', + 'qa/SKILL.md anti-refusal': 'judge', + 'cross-skill greptile consistency': 'judge', + 'ship/SKILL.md workflow': 'judge', + 'document-release/SKILL.md workflow': 'judge', + 'plan-ceo-review/SKILL.md modes': 'judge', + 'plan-eng-review/SKILL.md sections': 'judge', + 'plan-design-review/SKILL.md passes': 'judge', + 'design-review/SKILL.md fix loop': 'judge', + 'design-consultation/SKILL.md research': 'judge', + 'land-and-deploy/SKILL.md workflow': 'judge', + 'canary/SKILL.md monitoring loop': 'judge', + 'benchmark/SKILL.md perf collection': 'judge', + 'setup-deploy/SKILL.md platform setup': 'judge', + 'retro/SKILL.md instructions': 'judge', + 'qa-only/SKILL.md workflow': 'judge', + 'gstack-upgrade/SKILL.md upgrade flow': 'judge', + 'sync-gbrain/SKILL.md read-only readiness': 'judge', + 'voice directive tone': 'judge', }; /** @@ -1830,4 +1830,49 @@ export const E2E_KINDS: Record = { * deviation is acceptable product behavior. Keys equal the behavior ids of * E2E_KINDS; values are non-empty. */ -export const BEHAVIOR_WHY: Record = {}; +export const BEHAVIOR_WHY: Record = { + 'shared-libs-opportunity-judgment': + "Whether a candidate extraction is worth recommending is a judgment call; the read-only invariant stays a contract.", + 'review-design-lite': + "How many of the seven design-lite checklist items the live review flags varies run to run; the fake-engine rows it must carry stay strict.", + 'review-army-red-team': + "Whether the red-team lens surfaces on a small diff is a live model choice, not a contract.", + 'review-army-consensus': + "Multi-specialist agreement on the planted SQL finding is a quality benchmark that tolerates an occasional miss.", + 'review-army-simplification': + "Flagging the planted unnecessary structure is an advisory-lens quality judgment.", + 'review-army-simplification-precision': + "Staying silent on a lean diff is a false-flag noise benchmark; an occasional advisory is acceptable noise.", + 'office-hours-forcing-energy': + "The Q3 posture is scored by a live judge on generated prose; a single flat phrasing is tolerable.", + 'office-hours-builder-wildness': + "Builder-mode creativity is scored by a live judge on generated prose; one conservative riff is tolerable.", + 'office-hours-brain-writeback': + "The model's interpretation of the gbrain writeback instruction (page shape, tags) varies; no secret or safety step rides on it.", + 'office-hours-phase4-fork': + "Phase 4 asks the model to invent 2-3 architectures; surfacing the fork with its reasoning is open-ended generation.", + 'plan-ceo-review-expansion-energy': + "Expansion framing is scored by a live judge on generated proposals; one flat proposal set is tolerable.", + 'plan-ceo-review-format-mode': + "Mode-question wording (Completeness line vs kind note) is live formatting of one AskUserQuestion.", + 'plan-ceo-review-format-approach': + "Approach-menu Completeness wording is live formatting of one AskUserQuestion.", + 'plan-eng-review-format-coverage': + "Coverage-issue Completeness wording is live formatting of one AskUserQuestion.", + 'plan-eng-review-format-kind': + "Kind-note wording is live formatting of one AskUserQuestion.", + 'plan-ceo-review-prosons-cadence': + "Pros/Cons cadence on a hard-stop question is live formatting; either the escape or the full block is accepted.", + 'plan-review-prosons-format': + "The full Pros/Cons block (counts of pros and cons, labels) is live formatting of one question.", + 'plan-review-prosons-hardstop-neg': + "Not using the hard-stop escape on an ordinary decision is live formatting of one question.", + 'plan-review-prosons-neutral-neg': + "Avoiding neutral posture and naming a because-reason is live formatting of one question.", + 'design-html-slop-gate': + "How many scan passes the one-pass slop gate takes on a fake engine's fixed output is a judgment call.", + 'scrape-match-path': + "The /scrape fallback no longer prescribes the browser-skills match flow, so taking it is prompt compliance.", + 'scrape-prototype-path': + "The /scrape fallback no longer prescribes the prototype flow, so taking it is prompt compliance.", +}; From d6b3559b78a78d282f6b372f20246e02ebce2c8d Mon Sep 17 00:00:00 2001 From: garrytan Date: Tue, 29 Sep 2026 19:11:27 +0000 Subject: [PATCH 07/15] feat(evals): per-case pass rates with Wilson intervals, identity series and quarantine policy scripts/eval-flake-rank.ts becomes eval:pass-rates (eval:flake-rank stays an alias, and the legacy aggregate stays exported). It reads eval-store's trial-outcomes JSONL from the last N completed evals-periodic runs on this branch and main (gh, downloading only the trial-outcomes artifact, cached and size-capped, parsed as data), plus local eval dirs, and prints per-case per-trial pass rates with 95% Wilson intervals. A series is a case's own touchfiles minus GLOBAL_TOUCHFILES (caseSeriesIdentities, for the report job to stamp), per model, CLI version and policy version. Labels: INCONCLUSIVE, BROKEN, FLAKY, FAILING, PASSING. --backfill imports legacy slice artifacts as pre-policy trials (first attempt only, attributed by registry id, never guessed) for display only. --gate fails with ACTION REQUIRED on post-policy evidence only: drift below the quarantine entry rule, a rule case behaving like behavior, a one-sided Fisher drop against the previous identity (Holm-controlled), and quarantine entries that met their exit rule, expired after 8 weekly runs, broke the 10% tier cap or are invalid. CASE_QUARANTINE entries now carry a failureClass (detector, harness or model-latency); a product defect has no class and is never quarantined. The policy test pins EVAL_POLICY's approved constants. --- scripts/eval-flake-rank.ts | 620 ++++++++++++++++++++++++-- test/eval-flake-rank.test.ts | 309 ++++++++++++- test/helpers/periodic-exclude-data.ts | 14 +- test/periodic-exclude-policy.test.ts | 32 +- 4 files changed, 931 insertions(+), 44 deletions(-) diff --git a/scripts/eval-flake-rank.ts b/scripts/eval-flake-rank.ts index 3e6028a10..ca46a3eb8 100644 --- a/scripts/eval-flake-rank.ts +++ b/scripts/eval-flake-rank.ts @@ -1,29 +1,55 @@ #!/usr/bin/env bun /** - * eval-flake-rank — the flake-telemetry dial (WS1). + * eval-pass-rates (alias: eval-flake-rank) — per-case trial pass rates. * - * Aggregates per-test series across every FINALIZED eval-store run on this - * machine (default: ~/.gstack/projects//evals/, shard dirs included) - * plus the free suite's flake ledger, and ranks tests by flake signal: - * retried passes first (a test that needs attempt 2 to go green is the - * definition of a flake), then failure rate. + * Reads trial records (one JSONL line per trial: case, kind, trial, outcome, + * exit_reason, duration, cost, model, CLI version, series identity, run id, + * sha, policy_version) from the last N completed `evals-periodic.yml` runs on + * the current branch and `main` (downloading only each run's small + * `trial-outcomes` artifact through `gh`), plus any local eval dirs, and + * prints per-case per-trial pass rates with 95% Wilson intervals. * - * This is the readable dial behind two policies: - * - a flaky pass never blocks a merge, but it is recorded and RANKED here; - * - the required-check promotion (WS16) needs weeks of clean flake-rank, - * not vibes. + * A series is one case under one input identity: the case's own touchfiles + * minus GLOBAL_TOUCHFILES (`caseSeriesIdentities`), grouped by model and CLI + * version, per policy_version. A new identity starts a new series; earlier + * series stay visible. Only post-policy trials of the current series feed the + * labels and alarms. Legacy eval-store records (`--backfill`, `--dir`) are + * imported as pre-policy trials (first attempt only; a missing attempt means + * 1) and are display-only. + * + * Labels: INCONCLUSIVE (below the entry rule's minimum trials), BROKEN (latest run 0/n + * after a prior interval at or above the entry rate), FLAKY (failures and an + * interval straddling the entry rate), FAILING (interval below the entry + * rate), PASSING (otherwise). + * + * The weekly gate (`--gate`) exits non-zero with ACTION REQUIRED when a + * non-quarantined case meets the quarantine entry rule, a rule case behaves + * like a behavior case, a blocking case's current-identity rate is + * significantly below its previous identity (one-sided Fisher exact, + * Holm-controlled across cases), or a CASE_QUARANTINE entry has met its exit + * rule, expired, or pushed its tier over the cap. History that cannot be + * fetched fails the gate closed. * * Usage: - * bun run eval:flake-rank # project eval dir - * bun run eval:flake-rank --dir # e.g. downloaded CI artifacts - * bun run eval:flake-rank --json # machine-readable + * bun run eval:pass-rates # last 10 weekly runs, this branch + main + * bun run eval:pass-rates --case --runs 20 + * bun run eval:pass-rates --dir # local eval dirs / downloaded artifacts (repeatable) + * bun run eval:pass-rates --backfill # also import legacy slice artifacts, labeled pre-policy + * bun run eval:pass-rates --json | --gate */ import * as fs from 'node:fs'; +import * as os from 'node:os'; import * as path from 'node:path'; -import { getProjectEvalDir, isPartialEval, isFinalizedEvalResultFile, type EvalResult } from '../test/helpers/eval-store'; -import { evalEntryOutcome } from '../test/helpers/eval-store'; +import { spawnSync } from 'node:child_process'; +import { createHash } from 'node:crypto'; +import { isPartialEval, isFinalizedEvalResultFile, evalEntryOutcome, failureClassOf, parseTrialOutcomes, sanitizeTrialError, + TRIAL_OUTCOME_SCHEMA, type EvalCaseKind, type EvalResult, type TrialOutcomeRecord } from '../test/helpers/eval-store'; import { flakeLedgerPath, type FlakeLedgerEntry } from './test-free-shards'; +import { E2E_KINDS, E2E_TIERS, E2E_TOUCHFILES, GLOBAL_TOUCHFILES, LLM_JUDGE_TOUCHFILES } from '../test/helpers/touchfiles-data'; +import { CASE_QUARANTINE, EVAL_POLICY } from '../test/helpers/periodic-exclude-data'; +import { matchGlob } from '../test/helpers/test-selection'; +import { CASE_TEST_NAMES } from './test-paid-shards'; interface TestSeries { name: string; @@ -111,33 +137,558 @@ function readFreeLedger(): FlakeLedgerEntry[] { return out; } +// --- Trial records --- + +/** + * A trial record as pass-rates reads it: eval-store's trial-outcomes schema + * plus the series identity the report job stamps (caseSeriesIdentities). + * policy_version 0 marks a pre-policy (backfilled) record. + */ +export type TrialRecord = TrialOutcomeRecord & { series_identity?: string }; + +/** Per-file cap for downloaded artifacts: pass-rates parses data only, never executes it. */ +export const TRIAL_OUTCOMES_MAX_BYTES = 8 * 1024 * 1024; + +/** Every `trial-outcomes*.jsonl` file under a directory, size-capped, schema-validated by eval-store. */ +export function readTrialOutcomeDir(dir: string): { records: TrialRecord[]; errors: string[] } { + const records: TrialRecord[] = []; + const errors: string[] = []; + if (!fs.existsSync(dir)) return { records, errors }; + for (const name of fs.readdirSync(dir, { recursive: true }) as string[]) { + if (!/(^|\/)trial-outcomes[^/]*\.jsonl$/.test(name)) continue; + const full = path.join(dir, name); + const parsed = parseTrialOutcomes(fs.readFileSync(full, 'utf8'), { maxBytes: TRIAL_OUTCOMES_MAX_BYTES }); + records.push(...parsed.records.map(record => ({ + ...record, series_identity: typeof (record as TrialRecord).series_identity === 'string' + ? (record as TrialRecord).series_identity!.slice(0, 64) : undefined }))); + errors.push(...parsed.errors.map(error => `${full}: ${error}`)); + } + return { records, errors }; +} + +// --- Registry attribution and series identity --- + +export interface Registry { + kinds: Record; + tiers: Record; + touchfiles: Record; + judgeTouchfiles: Record; + globals: readonly string[]; + testNames: Record; +} + +export const LIVE_REGISTRY: Registry = { + kinds: E2E_KINDS, tiers: E2E_TIERS, touchfiles: E2E_TOUCHFILES, judgeTouchfiles: LLM_JUDGE_TOUCHFILES, + globals: GLOBAL_TOUCHFILES, testNames: CASE_TEST_NAMES, +}; + +/** A case's tier: its E2E_TIERS value, or 'judge' for an LLM-judge entry. */ +export function caseTier(id: string, registry: Registry = LIVE_REGISTRY): string { + return registry.tiers[id] ?? (id in registry.judgeTouchfiles ? 'judge' : 'unknown'); +} + +/** + * Attribute a legacy eval-store record to a registry id: the case-shard slug + * suffix (`--`), the recorded name, a CASE_TEST_NAMES label, or the + * only id its shard file registers. Anything else is unattributed (null). + */ +export function attributeLegacyRecord(name: string, shard: string | undefined, registry: Registry = LIVE_REGISTRY): string | null { + const known = (id: string) => id in registry.kinds; + const [slugFile, slugCase] = (shard ?? '').split('--'); + if (slugCase && known(slugCase)) return slugCase; + if (known(name)) return name; + const labeled = Object.entries(registry.testNames).find(([, label]) => label === name)?.[0]; + if (labeled && known(labeled)) return labeled; + if (slugFile) { + const file = `test/${slugFile}.test.ts`; + const owners = Object.keys(registry.touchfiles).filter(id => registry.touchfiles[id]!.includes(file)); + if (owners.length === 1 && known(owners[0]!)) return owners[0]!; + } + return null; +} + +/** + * Series identity per case: a hash of the git blob ids of the files matching + * the case's own touchfiles, excluding GLOBAL_TOUCHFILES (harness edits are + * markers, not new series). The report job stamps this on every trial record. + */ +export function caseSeriesIdentities(ids: string[], root: string, registry: Registry = LIVE_REGISTRY): Record { + const listed = spawnSync('git', ['ls-files', '-s'], { cwd: root, encoding: 'utf8', timeout: 20_000, maxBuffer: 64 * 1024 * 1024 }); + if (listed.status !== 0) throw new Error(`git ls-files failed: ${listed.stderr}`); + const blobs = listed.stdout.split('\n').filter(Boolean).map(line => { + const [meta, file] = line.split('\t'); + return { file: file!, blob: meta!.split(' ')[1]! }; + }).filter(entry => !registry.globals.some(pattern => matchGlob(entry.file, pattern))); + return Object.fromEntries(ids.map(id => { + const patterns = registry.touchfiles[id] ?? registry.judgeTouchfiles[id] ?? []; + const lines = blobs.filter(entry => patterns.some(pattern => matchGlob(entry.file, pattern))) + .map(entry => `${entry.file} ${entry.blob}`).sort(); + return [id, createHash('sha256').update(`${id}\n${lines.join('\n')}`).digest('hex').slice(0, 16)]; + })); +} + +/** + * Import legacy eval-store result files as pre-policy trials (policy_version + * 0, source 'backfill'): first attempt only (a missing attempt means 1), + * attributed by registry id, never guessed. A manual-review acceptance carries + * no automated verdict: it is counted and shown, never scored. Without a CI + * run, each local result file is its own run. + */ +export function backfillEvalFiles(files: string[], run?: { run_id: string; sha?: string; timestamp?: string }, + registry: Registry = LIVE_REGISTRY): { records: TrialRecord[]; unattributed: string[]; manualReviews: string[] } { + const records: TrialRecord[] = []; + const unattributed = new Set(); + const manualReviews: string[] = []; + for (const file of files) { + let result: EvalResult & { shard?: string; claude_cli_version?: string }; + try { result = JSON.parse(fs.readFileSync(file, 'utf8')); } catch { continue; } + if (isPartialEval(result, file) || !Array.isArray(result.tests)) continue; + const seen = new Set(); + for (const entry of result.tests) { + if ((entry.attempt ?? 1) !== 1 || seen.has(entry.name)) continue; + seen.add(entry.name); + const id = attributeLegacyRecord(entry.name, result.shard, registry); + if (!id) { unattributed.add(entry.name); continue; } + const outcome = evalEntryOutcome(entry); + if (outcome === 'manual-review') { manualReviews.push(id); continue; } + records.push({ + schema: TRIAL_OUTCOME_SCHEMA, case: id, + file: result.shard ? `test/${result.shard.split('--')[0]}.test.ts` : 'unknown', + tier: caseTier(id, registry), kind: registry.kinds[id]!, trial: 1, panel: { n: 1, k: 1 }, attempt: 1, + outcome, ...(outcome === 'failed' ? { failure_class: failureClassOf(entry) } : {}), + exit_reason: entry.exit_reason, error: sanitizeTrialError(entry.error), + duration_ms: Math.max(0, entry.duration_ms || 0), cost_usd: Math.max(0, entry.cost_usd || 0), + model: entry.model, cli_version: result.claude_cli_version, policy_version: 0, quarantined: false, + execution: entry.execution === 'reused' ? 'reused' : 'executed', source: 'backfill', + run_id: run?.run_id ?? `local:${file}`, sha: run?.sha ?? result.git_sha, recorded_at: run?.timestamp ?? result.timestamp, + }); + } + } + return { records, unattributed: [...unattributed].sort(), manualReviews }; +} + +// --- Statistics --- + +/** 95% Wilson score interval for k successes in n trials. */ +export function wilsonInterval(k: number, n: number, z = 1.96): { lo: number; hi: number } { + if (n <= 0) return { lo: 0, hi: 1 }; + const p = k / n, z2 = z * z, denom = 1 + z2 / n; + const center = (p + z2 / (2 * n)) / denom; + const half = (z * Math.sqrt(p * (1 - p) / n + z2 / (4 * n * n))) / denom; + return { lo: Math.max(0, center - half), hi: k === n ? 1 : Math.min(1, center + half) }; +} + +function logChoose(n: number, k: number): number { + let sum = 0; + for (let i = 1; i <= k; i++) sum += Math.log(n - k + i) - Math.log(i); + return sum; +} + +/** + * One-sided Fisher exact p-value that the CURRENT pass rate is below the + * PREVIOUS one: P(X <= curPass) under the hypergeometric null with the + * observed margins. + */ +export function fisherOneSidedLower(curPass: number, curN: number, prevPass: number, prevN: number): number { + const passes = curPass + prevPass, total = curN + prevN; + const denom = logChoose(total, passes); + let p = 0; + for (let x = Math.max(0, passes - prevN); x <= curPass; x++) p += Math.exp(logChoose(curN, x) + logChoose(prevN, passes - x) - denom); + return Math.min(1, p); +} + +/** Holm step-down: the indices whose p-values are rejected at family-wise alpha. */ +export function holmRejections(pValues: number[], alpha: number): Set { + const order = pValues.map((p, index) => ({ p, index })).sort((a, b) => a.p - b.p); + const rejected = new Set(); + for (let rank = 0; rank < order.length; rank++) { + if (order[rank]!.p > alpha / (order.length - rank)) break; + rejected.add(order[rank]!.index); + } + return rejected; +} + +// --- Analysis --- + +export type PassRateLabel = 'INCONCLUSIVE' | 'BROKEN' | 'FLAKY' | 'FAILING' | 'PASSING'; +export type AlarmKind = 'drift' | 'rule-as-behavior' | 'regression' | 'quarantine-exit' | 'quarantine-expired' + | 'quarantine-cap' | 'quarantine-invalid'; + +/** The EVAL_POLICY fields pass-rates reads (structural, so tests can vary them). */ +export interface PassRatePolicy { + version: number; + quarantine: { entry: { rate: number; minTrials: number }; exit: { rate: number; minTrials: number }; capFraction: number; expiryWeeklyRuns: number }; + drift: { fisherAlpha: number; fisherMinPerSide: number }; +} + +export type QuarantineEntry = (typeof CASE_QUARANTINE)[string]; + +/** Tiers whose cases block a lane; quarantine applies only to them. */ +export const BLOCKING_TIERS: readonly string[] = ['gate', 'periodic']; +const QUARANTINE_FAILURE_CLASSES: readonly string[] = ['detector', 'harness', 'model-latency']; + +export interface SeriesStats { + key: string; + identity: string; + model: string; + cli: string; + policyVersion: number; + passes: number; + /** Scored trials: passed + failed (skipped trials carry no verdict). */ + trials: number; + infra: number; + interval: { lo: number; hi: number }; + firstSeen: string; + lastSeen: string; + runs: string[]; +} + +export interface CasePassRate { + case: string; + kind: EvalCaseKind; + tier: string; + quarantined: boolean; + label: PassRateLabel; + /** Manual-review acceptances: visible, never scored. */ + manualReviews: number; + current: SeriesStats | null; + previous: SeriesStats | null; + prePolicy: SeriesStats | null; + series: SeriesStats[]; + latestRun: { runId: string; passes: number; trials: number } | null; +} + +export interface Alarm { kind: AlarmKind; case: string; message: string } + +export interface PassRateReport { + policyVersion: number; + cases: CasePassRate[]; + alarms: Alarm[]; + postPolicyTrials: number; + prePolicyTrials: number; + unattributed: string[]; + errors: string[]; +} + +export interface AnalyzeOptions { + registry?: Registry; + quarantine?: Record; + policy?: PassRatePolicy; + /** Completed weekly-run timestamps in the window, for quarantine expiry. */ + weeklyRuns?: string[]; + now?: number; + unattributed?: string[]; + errors?: string[]; + manualReviews?: string[]; +} + +const at = (record: TrialRecord) => record.recorded_at ?? ''; +const runOf = (record: TrialRecord) => `${record.run_id ?? record.sha ?? 'local'}#${record.attempt}`; + +function seriesStats(key: string, records: TrialRecord[]): SeriesStats { + const scored = records.filter(record => record.outcome !== 'skipped'); + const passes = scored.filter(record => record.outcome === 'passed').length; + const times = records.map(at).sort(); + const first = records[0]!; + return { + key, identity: first.series_identity ?? 'unknown', model: first.model ?? 'unknown', cli: first.cli_version ?? 'unknown', + policyVersion: first.policy_version, passes, trials: scored.length, + infra: scored.filter(record => record.outcome === 'failed' && record.failure_class === 'infra').length, + interval: wilsonInterval(passes, scored.length), firstSeen: times[0] ?? '', lastSeen: times[times.length - 1] ?? '', + runs: [...new Set(records.map(runOf))], + }; +} + +/** Weekly runs completed after an entry's enteredAt; offline, whole weeks elapsed. */ +export function quarantineRunsSince(enteredAt: string, weeklyRuns: string[] | undefined, now: number): number { + const entered = Date.parse(enteredAt); + if (!Number.isFinite(entered)) return Number.POSITIVE_INFINITY; + if (weeklyRuns && weeklyRuns.length) return weeklyRuns.filter(time => Date.parse(time) > entered).length; + return Math.floor((now - entered) / (7 * 86_400_000)); +} + +/** + * Static CASE_QUARANTINE problems, shared by the free policy test and the + * weekly gate: an id that is not a blocking-tier E2E case, a missing field, + * a failure class outside detector / harness / model-latency (a product + * defect is fixed or named, never quarantined), a malformed or future date, + * and a tier over its cap. + */ +export function quarantinePolicyProblems(quarantine: Record, + registry: Registry = LIVE_REGISTRY, policy: PassRatePolicy = EVAL_POLICY, now = Date.now()): Alarm[] { + const problems: Alarm[] = []; + const invalid = (id: string, message: string) => problems.push({ kind: 'quarantine-invalid', case: id, message: `${id}: ${message}` }); + const perTier = new Map(); + for (const [id, entry] of Object.entries(quarantine)) { + const tier = registry.tiers[id]; + if (!tier || !(id in registry.kinds)) { invalid(id, 'CASE_QUARANTINE names no registered E2E case'); continue; } + if (!BLOCKING_TIERS.includes(tier)) invalid(id, `tier ${tier} is not blocking; only ${BLOCKING_TIERS.join(' and ')} cases are quarantined`); + for (const field of ['reason', 'failureClass', 'tracking', 'owner', 'enteredAt', 'exit'] as const) { + if (typeof entry[field] !== 'string' || !entry[field].trim()) invalid(id, `missing ${field}`); + } + if (typeof entry.reason === 'string' && entry.reason.trim().length < 40) invalid(id, 'reason must be a written diagnosis (at least 40 characters)'); + if (!QUARANTINE_FAILURE_CLASSES.includes(entry.failureClass)) { + invalid(id, `failureClass ${JSON.stringify(entry.failureClass)} is not ${QUARANTINE_FAILURE_CLASSES.join(', ')}; a product defect is fixed or named as a red, never quarantined`); + } + const entered = Date.parse(entry.enteredAt); + if (!/^\d{4}-\d{2}-\d{2}$/.test(entry.enteredAt ?? '') || !Number.isFinite(entered)) invalid(id, 'enteredAt must be YYYY-MM-DD'); + else if (entered > now) invalid(id, 'enteredAt is in the future'); + perTier.set(tier, (perTier.get(tier) ?? 0) + 1); + } + for (const [tier, count] of perTier) { + const size = Object.values(registry.tiers).filter(value => value === tier).length; + const cap = Math.floor(size * policy.quarantine.capFraction); + if (count > cap) problems.push({ kind: 'quarantine-cap', case: tier, + message: `${count} quarantined ${tier} cases exceed the ${pct(policy.quarantine.capFraction)} cap (${cap} of ${size})` }); + } + return problems; +} + +export function analyzePassRates(records: TrialRecord[], options: AnalyzeOptions = {}): PassRateReport { + const registry = options.registry ?? LIVE_REGISTRY; + const quarantine = options.quarantine ?? CASE_QUARANTINE; + const policy = options.policy ?? EVAL_POLICY; + const now = options.now ?? Date.now(); + const byCase = new Map(); + for (const record of records) { + const list = byCase.get(record.case) ?? []; + list.push(record); + byCase.set(record.case, list); + } + for (const id of options.manualReviews ?? []) if (!byCase.has(id)) byCase.set(id, []); + const cases: CasePassRate[] = []; + for (const [id, list] of [...byCase].sort(([a], [b]) => a.localeCompare(b))) { + list.sort((a, b) => at(a).localeCompare(at(b)) || runOf(a).localeCompare(runOf(b)) || a.trial - b.trial); + const groups = new Map(); + for (const record of list) { + const key = record.policy_version === 0 ? 'pre-policy' + : [record.series_identity ?? 'unknown', record.model ?? 'unknown', record.cli_version ?? 'unknown', `v${record.policy_version}`].join('|'); + const group = groups.get(key) ?? []; + group.push(record); + groups.set(key, group); + } + const series = [...groups].map(([key, group]) => seriesStats(key, group)) + .sort((a, b) => a.lastSeen.localeCompare(b.lastSeen)); + const post = series.filter(entry => entry.policyVersion !== 0); + const current = post[post.length - 1] ?? null; + const previous = post[post.length - 2] ?? null; + const scored = current ? groups.get(current.key)!.filter(record => record.outcome !== 'skipped') : []; + const latestRun = scored.length ? runOf(scored[scored.length - 1]!) : null; + const latest = scored.filter(record => runOf(record) === latestRun); + const prior = scored.filter(record => runOf(record) !== latestRun); + const priorPasses = prior.filter(record => record.outcome === 'passed').length; + const entryRate = policy.quarantine.entry.rate; + let label: PassRateLabel; + if (latest.length > 0 && latest.every(record => record.outcome === 'failed') + && prior.length > 0 && wilsonInterval(priorPasses, prior.length).lo >= entryRate) label = 'BROKEN'; + else if (!current || current.trials < policy.quarantine.entry.minTrials) label = 'INCONCLUSIVE'; + else if (current.interval.hi < entryRate) label = 'FAILING'; + else if (current.passes < current.trials && current.interval.lo < entryRate) label = 'FLAKY'; + else label = 'PASSING'; + cases.push({ + case: id, kind: registry.kinds[id] ?? list[0]!.kind, tier: caseTier(id, registry), + quarantined: id in quarantine, label, current, previous, + manualReviews: (options.manualReviews ?? []).filter(name => name === id).length, + prePolicy: series.find(entry => entry.policyVersion === 0) ?? null, series, + latestRun: latestRun ? { runId: latestRun, passes: latest.filter(record => record.outcome === 'passed').length, trials: latest.length } : null, + }); + } + + const alarms: Alarm[] = []; + const rate = (stats: SeriesStats) => stats.passes / stats.trials; + for (const entry of cases) { + const current = entry.current; + if (!current) continue; + const below = current.trials >= policy.quarantine.entry.minTrials && rate(current) < policy.quarantine.entry.rate; + if (below && !entry.quarantined && BLOCKING_TIERS.includes(entry.tier)) alarms.push({ kind: 'drift', case: entry.case, + message: `${entry.case} passes ${current.passes}/${current.trials} (below ${pct(policy.quarantine.entry.rate)} over >= ${policy.quarantine.entry.minTrials} trials): fix it, or propose a CASE_QUARANTINE entry with a written diagnosis (product defects are never quarantined)` }); + if (below && entry.kind === 'rule') alarms.push({ kind: 'rule-as-behavior', case: entry.case, + message: `${entry.case}: rule case behaving like behavior (${current.passes}/${current.trials}): fix or reclassify` }); + if (entry.quarantined && current.trials >= policy.quarantine.exit.minTrials && rate(current) >= policy.quarantine.exit.rate) { + alarms.push({ kind: 'quarantine-exit', case: entry.case, + message: `${entry.case} passes ${current.passes}/${current.trials} (>= ${pct(policy.quarantine.exit.rate)}): remove its CASE_QUARANTINE entry` }); + } + } + const tested = cases.filter(entry => BLOCKING_TIERS.includes(entry.tier) && entry.current && entry.previous + && entry.current.trials >= policy.drift.fisherMinPerSide && entry.previous.trials >= policy.drift.fisherMinPerSide); + const pValues = tested.map(entry => fisherOneSidedLower(entry.current!.passes, entry.current!.trials, entry.previous!.passes, entry.previous!.trials)); + for (const index of holmRejections(pValues, policy.drift.fisherAlpha)) { + const entry = tested[index]!; + alarms.push({ kind: 'regression', case: entry.case, + message: `${entry.case}: current identity ${entry.current!.passes}/${entry.current!.trials} is significantly below the previous ${entry.previous!.passes}/${entry.previous!.trials} (one-sided Fisher p=${pValues[index]!.toFixed(4)}, Holm over ${tested.length} cases)` }); + } + for (const [id, entry] of Object.entries(quarantine)) { + const runs = quarantineRunsSince(entry.enteredAt, options.weeklyRuns, now); + if (runs >= policy.quarantine.expiryWeeklyRuns) alarms.push({ kind: 'quarantine-expired', case: id, + message: `${id}: entered ${runs} weekly runs ago (limit ${policy.quarantine.expiryWeeklyRuns}): fix it, name it as a red, or re-diagnose with fresh evidence` }); + } + alarms.push(...quarantinePolicyProblems(quarantine, registry, policy, now)); + + const post = records.filter(record => record.policy_version !== 0).length; + return { policyVersion: policy.version, cases, alarms, postPolicyTrials: post, prePolicyTrials: records.length - post, + unattributed: options.unattributed ?? [], errors: options.errors ?? [] }; +} + +function pct(value: number): string { return `${Math.round(value * 1000) / 10}%`; } + +function formatStats(stats: SeriesStats | null): string { + if (!stats) return '-'; + return `${stats.passes}/${stats.trials} [${pct(stats.interval.lo)}–${pct(stats.interval.hi)}]${stats.infra ? ` (${stats.infra} infra)` : ''}`; +} + +export function formatPassRates(report: PassRateReport, options: { caseFilter?: string } = {}): string { + const lines: string[] = []; + lines.push(`pass-rates: policy v${report.policyVersion}, ${report.postPolicyTrials} post-policy trial(s), ${report.prePolicyTrials} pre-policy (display only)`); + if (report.postPolicyTrials === 0) lines.push(' no post-policy trials yet: every series starts INCONCLUSIVE'); + const cases = report.cases.filter(entry => !options.caseFilter || entry.case === options.caseFilter); + lines.push(' label kind tier current series pre-policy manual case'); + for (const entry of cases) { + const group = entry.current ? ` ${entry.current.model} / ${entry.current.cli}` : ''; + const reset = entry.previous ? ' (baseline reset)' : ''; + lines.push(` ${entry.label.padEnd(12)} ${entry.kind.padEnd(8)} ${entry.tier.padEnd(8)} ${formatStats(entry.current).padEnd(29)} ` + + `${formatStats(entry.prePolicy).padEnd(18)} ${String(entry.manualReviews).padStart(6)} ${entry.case}${entry.quarantined ? ' [quarantined]' : ''}${group}${reset}`); + } + if (report.unattributed.length) lines.push(` unattributed records (${report.unattributed.length}, never guessed): ${report.unattributed.slice(0, 20).join(', ')}`); + if (report.errors.length) lines.push(` rejected ${report.errors.length} invalid trial line(s): ${report.errors.slice(0, 5).join('; ')}`); + if (report.alarms.length) { + lines.push(`ACTION REQUIRED (${report.alarms.length}):`); + for (const alarm of report.alarms) lines.push(` [${alarm.kind}] ${alarm.message}`); + } + return lines.join('\n'); +} + +// --- GitHub history --- + +export interface WeeklyRun { id: number; attempt: number; sha: string; branch: string; createdAt: string } +export interface RunArtifact { id: number; name: string; size: number } + +/** The GitHub calls pass-rates makes; injectable so the free tests never touch the network. */ +export interface HistoryFetcher { + listRuns(repo: string, workflow: string, branch: string, limit: number): WeeklyRun[]; + listArtifacts(repo: string, runId: number): RunArtifact[]; + downloadZip(repo: string, artifactId: number, destination: string): void; +} + +function gh(args: string[]): Buffer { + const result = spawnSync('gh', args, { timeout: 300_000, maxBuffer: 256 * 1024 * 1024 }); + if (result.status !== 0) throw new Error(`gh ${args.slice(0, 2).join(' ')} failed: ${String(result.stderr || result.error || '').trim()}`); + return result.stdout; +} + +const jsonLines = (buffer: Buffer): T[] => buffer.toString('utf8').split('\n').filter(Boolean).map(line => JSON.parse(line) as T); + +export const GH_HISTORY: HistoryFetcher = { + listRuns: (repo, workflow, branch, limit) => jsonLines(gh(['api', + `repos/${repo}/actions/workflows/${workflow}/runs?branch=${encodeURIComponent(branch)}&status=completed&per_page=${limit}`, + '--jq', '.workflow_runs[] | {id, attempt: .run_attempt, sha: .head_sha, branch: .head_branch, createdAt: .created_at}'])), + listArtifacts: (repo, runId) => jsonLines(gh(['api', `repos/${repo}/actions/runs/${runId}/artifacts?per_page=100`, + '--paginate', '--jq', '.artifacts[] | select(.expired | not) | {id, name, size: .size_in_bytes}'])), + downloadZip: (repo, artifactId, destination) => fs.writeFileSync(destination, gh(['api', `repos/${repo}/actions/artifacts/${artifactId}/zip`])), +}; + +/** The last `limit` completed runs of `workflow` on each branch, newest first, deduplicated. */ +export function listWeeklyRuns(opts: { repo: string; workflow: string; branches: string[]; limit: number; fetcher?: HistoryFetcher }): WeeklyRun[] { + const fetcher = opts.fetcher ?? GH_HISTORY; + const runs = new Map(); + for (const branch of opts.branches) for (const run of fetcher.listRuns(opts.repo, opts.workflow, branch, opts.limit)) runs.set(run.id, run); + return [...runs.values()].sort((a, b) => b.createdAt.localeCompare(a.createdAt)); +} + +/** + * Download the artifacts of one run whose names match into a per-run cache + * directory (reused on later calls) and return the extracted directories. + * Oversized or oddly named artifacts are skipped: downloads are data only. + */ +export function downloadRunArtifacts(opts: { repo: string; run: WeeklyRun; match: (name: string) => boolean; cacheDir: string; + fetcher?: HistoryFetcher; maxBytes?: number }): string[] { + const fetcher = opts.fetcher ?? GH_HISTORY; + const dirs: string[] = []; + for (const artifact of fetcher.listArtifacts(opts.repo, opts.run.id)) { + if (!opts.match(artifact.name) || !/^[A-Za-z0-9._-]+$/.test(artifact.name)) continue; + if (artifact.size > (opts.maxBytes ?? TRIAL_OUTCOMES_MAX_BYTES)) continue; + const dir = path.join(opts.cacheDir, `${opts.run.id}`, artifact.name); + if (!fs.existsSync(path.join(dir, '.complete'))) { + fs.rmSync(dir, { recursive: true, force: true }); + fs.mkdirSync(dir, { recursive: true }); + const zip = path.join(dir, 'artifact.zip'); + fetcher.downloadZip(opts.repo, artifact.id, zip); + const unzip = spawnSync('unzip', ['-o', '-q', zip, '-d', dir], { timeout: 120_000 }); + if (unzip.status !== 0) throw new Error(`unzip failed for ${artifact.name}: ${String(unzip.stderr || unzip.error || '')}`); + fs.rmSync(zip, { force: true }); + fs.writeFileSync(path.join(dir, '.complete'), ''); + } + dirs.push(dir); + } + return dirs; +} + +function gitOutput(args: string[]): string | null { + const result = spawnSync('git', args, { encoding: 'utf8', timeout: 5_000 }); + return result.status === 0 ? result.stdout.trim() : null; +} + +function repoSlug(): string { + const url = gitOutput(['remote', 'get-url', 'origin']) ?? ''; + return url.match(/[:/]([^/:]+\/[^/]+?)(?:\.git)?$/)?.[1] ?? 'garrytan/gstack'; +} + if (import.meta.main) { const argv = process.argv.slice(2); - const dirFlag = argv.indexOf('--dir'); - const dir = dirFlag !== -1 ? argv[dirFlag + 1] : getProjectEvalDir(); + const flag = (name: string) => { const index = argv.indexOf(name); return index === -1 ? undefined : argv[index + 1]; }; + const dirs = argv.flatMap((arg, index) => arg === '--dir' && argv[index + 1] ? [argv[index + 1]!] : []); const asJson = argv.includes('--json'); - const sinceFlag = argv.indexOf('--since-days'); - const sinceDays = sinceFlag !== -1 ? Number(argv[sinceFlag + 1]) || 60 : 60; + const gate = argv.includes('--gate'); + const backfill = argv.includes('--backfill'); + const caseFilter = flag('--case'); + const runsLimit = Number(flag('--runs')) || 10; + const sinceDays = Number(flag('--since-days')) || 60; + const repo = flag('--repo') ?? repoSlug(); + const workflow = flag('--workflow') ?? 'evals-periodic.yml'; + const branch = flag('--branch') ?? gitOutput(['rev-parse', '--abbrev-ref', 'HEAD']) ?? 'main'; - const files = collectEvalFiles(dir, sinceDays); - const series = [...aggregate(files).values()] - .sort((a, b) => b.retriedPasses - a.retriedPasses || (b.fails / Math.max(1, b.runs)) - (a.fails / Math.max(1, a.runs))); - const ledger = readFreeLedger(); + const records: TrialRecord[] = []; + const unattributed = new Set(); + const errors: string[] = []; + let historyError: string | null = null; + let weeklyRuns: string[] | undefined; - if (asJson) { - console.log(JSON.stringify({ dir, runsScanned: files.length, tests: series, freeLedger: ledger }, null, 2)); + const manualReviews: string[] = []; + const importDir = (dir: string, run: { run_id: string; sha?: string; timestamp?: string } | undefined, legacyDays: number) => { + const trials = readTrialOutcomeDir(dir); + records.push(...trials.records); + errors.push(...trials.errors); + const legacy = backfillEvalFiles(collectEvalFiles(dir, legacyDays), run); + records.push(...legacy.records); + manualReviews.push(...legacy.manualReviews); + legacy.unattributed.forEach(name => unattributed.add(name)); + }; + + if (dirs.length) { + for (const dir of dirs) importDir(dir, undefined, sinceDays); } else { - console.log(`flake-rank: ${files.length} finalized run file(s) under ${dir}`); - const flaky = series.filter((s) => s.retriedPasses > 0 || s.fails > 0 || s.manualAccepted > 0); - if (flaky.length === 0) { - console.log(' no retried passes and no failures recorded — clean series'); - } else { - console.log(' retries fails/runs manual avg-dur test'); - for (const s of flaky.slice(0, 30)) { - console.log(` ${String(s.retriedPasses).padStart(7)} ${String(s.fails).padStart(5)}/${String(s.runs).padEnd(6)} ` - + `${String(s.manualAccepted).padStart(6)} ${Math.round(s.totalDurationMs / s.totalAttempts / 1000).toString().padStart(5)}s ${s.name}`); + try { + const runs = listWeeklyRuns({ repo, workflow, branches: [...new Set([branch, 'main'])], limit: runsLimit }); + weeklyRuns = runs.map(run => run.createdAt); + const cacheDir = path.join(os.homedir(), '.gstack', 'eval-pass-rates-cache', repo.replace('/', '-')); + const match = backfill + ? (name: string) => name.startsWith('trial-outcomes') || /^(paid-slice-\d+|gate-census-\d+)$/.test(name) + : (name: string) => name.startsWith('trial-outcomes'); + for (const run of runs) { + const dirsForRun = downloadRunArtifacts({ repo, run, match, cacheDir, maxBytes: backfill ? 64 * 1024 * 1024 : undefined }); + for (const dir of dirsForRun) importDir(dir, { run_id: `${run.id}`, sha: run.sha, timestamp: run.createdAt }, 3650); } + } catch (error) { + historyError = error instanceof Error ? error.message : String(error); } + } + + const report = analyzePassRates(records, { weeklyRuns, unattributed: [...unattributed].sort(), errors, manualReviews }); + const ledger = readFreeLedger(); + if (asJson) { + console.log(JSON.stringify({ repo, workflow, branch, dirs, historyError, ...report, freeLedger: ledger }, null, 2)); + } else { + if (historyError) console.log(`pass-rates: history unavailable (${historyError}); every label below is INCONCLUSIVE`); + console.log(formatPassRates(report, { caseFilter })); if (ledger.length > 0) { const byFile = new Map(); for (const e of ledger) byFile.set(e.file, (byFile.get(e.file) ?? 0) + 1); @@ -147,4 +698,5 @@ if (import.meta.main) { } } } + if (gate && (historyError || report.alarms.length)) process.exit(1); } diff --git a/test/eval-flake-rank.test.ts b/test/eval-flake-rank.test.ts index 198a64d54..0eaa1ddab 100644 --- a/test/eval-flake-rank.test.ts +++ b/test/eval-flake-rank.test.ts @@ -13,6 +13,13 @@ import * as path from 'node:path'; import { spawnSync } from 'node:child_process'; import { aggregate, collectEvalFiles } from '../scripts/eval-flake-rank'; import { manualReviewFixture } from './helpers/manual-judge-review-fixture'; +import { + analyzePassRates, attributeLegacyRecord, backfillEvalFiles, caseSeriesIdentities, downloadRunArtifacts, fisherOneSidedLower, + formatPassRates, holmRejections, listWeeklyRuns, quarantinePolicyProblems, quarantineRunsSince, readTrialOutcomeDir, + wilsonInterval, type HistoryFetcher, type PassRatePolicy, type QuarantineEntry, type Registry, type TrialRecord, +} from '../scripts/eval-flake-rank'; +import { EVAL_POLICY } from './helpers/periodic-exclude-data'; +import { TRIAL_OUTCOME_SCHEMA, formatTrialOutcomes } from './helpers/eval-store'; const entry = (name: string, passed: boolean, attempt: number) => ({ name, suite: 's', tier: 'e2e', passed, attempt, duration_ms: 1000, cost_usd: 0.1, @@ -39,9 +46,10 @@ describe('eval-flake-rank aggregate', () => { const display = spawnSync(process.execPath, [path.resolve(import.meta.dir, '../scripts/eval-flake-rank.ts'), '--dir', dir], { encoding: 'utf8', timeout: 10_000 }); expect(display.status, display.stderr).toBe(0); - expect(display.stdout).toContain('fails/runs manual'); - expect(display.stdout).toContain('0/1'); - expect(display.stdout).toContain(manual.name); + // pass-rates view: the prior automated pass is the one scored pre-policy + // trial; the manual acceptance is counted in its own column, never scored. + expect(display.stdout).toContain('pre-policy manual case'); + expect(display.stdout).toMatch(new RegExp(`1/1 \\[[^\\]]+\\]\\s+1 ${manual.name.replace(/[.*+?^${}()|[\]\\/]/g, '\\$&')}`)); fs.writeFileSync(path.join(dir, 'invalid-retry.json'), run([ { ...ordinary, attempt: 1 }, { ...manual, attempt: 2 }, ])); @@ -97,3 +105,298 @@ describe('eval-flake-rank aggregate', () => { fs.rmSync(dir, { recursive: true, force: true }); }); }); + +// --- pass-rates --- + +const registry: Registry = { + kinds: { 'rule-a': 'rule', 'beh-b': 'behavior', 'gate-c': 'rule', 'mar-d': 'rule', 'judge one': 'judge', + ...Object.fromEntries(Array.from({ length: 18 }, (_, i) => [`filler-${i}`, 'rule'])) }, + tiers: { 'rule-a': 'periodic', 'beh-b': 'periodic', 'gate-c': 'gate', 'mar-d': 'marathon', + ...Object.fromEntries(Array.from({ length: 18 }, (_, i) => [`filler-${i}`, i < 9 ? 'gate' : 'periodic'])) }, + touchfiles: { 'rule-a': ['test/skill-e2e-a.test.ts', 'a/**'], 'beh-b': ['test/skill-e2e-b.test.ts', 'b/**'], + 'gate-c': ['test/skill-e2e-shared.test.ts'], 'mar-d': ['test/skill-e2e-shared.test.ts'] }, + judgeTouchfiles: { 'judge one': ['j/SKILL.md'] }, + globals: ['harness/**'], + testNames: { 'gate-c': '/gate c labeled' }, +}; + +let clock = 0; +function trial(id: string, outcome: 'passed' | 'failed' | 'skipped', extra: Partial = {}): TrialRecord { + clock += 1; + return { + schema: TRIAL_OUTCOME_SCHEMA, case: id, file: 'test/x.test.ts', tier: registry.tiers[id] ?? 'judge', + kind: registry.kinds[id]!, trial: 1, panel: { n: 1, k: 1 }, attempt: 1, outcome, + ...(outcome === 'failed' ? { failure_class: 'assertion' as const } : {}), + duration_ms: 1, cost_usd: 0, model: 'model-x', cli_version: '2.1.284', policy_version: 1, quarantined: false, + execution: 'executed', source: 'shard', run_id: `run-${clock}`, recorded_at: new Date(Date.UTC(2026, 9, 1) + clock * 60_000).toISOString(), + series_identity: 'id-1', ...extra, + }; +} +const many = (id: string, passes: number, fails: number, extra: Partial = {}) => + [...Array.from({ length: passes }, () => trial(id, 'passed', extra)), ...Array.from({ length: fails }, () => trial(id, 'failed', extra))]; +const analyze = (records: TrialRecord[], quarantine: Record = {}, extra = {}) => + analyzePassRates(records, { registry, quarantine, now: Date.UTC(2026, 9, 2), ...extra }); +const qEntry = (overrides: Partial = {}): QuarantineEntry => ({ + reason: 'Detector graded the posture wording; 7 of 10 fresh trials failed only the regex, transcripts attached.', + failureClass: 'detector', tracking: 'TODOS.md "x"', owner: 'garrytan', enteredAt: '2026-09-29', + exit: '>= 97% over >= 10 trials on the current identity', ...overrides, +}); + +describe('pass-rates statistics', () => { + test('Wilson bounds match the documented policy arithmetic', () => { + expect(wilsonInterval(10, 10).lo).toBeCloseTo(0.7225, 4); + expect(wilsonInterval(6, 6).lo).toBeCloseTo(0.6097, 4); + expect(wilsonInterval(125, 125).lo).toBeCloseTo(0.9702, 4); + expect(wilsonInterval(10, 10).hi).toBe(1); + expect(wilsonInterval(0, 0)).toEqual({ lo: 0, hi: 1 }); + const mid = wilsonInterval(7, 10); + expect(mid.lo).toBeGreaterThan(0.39); expect(mid.hi).toBeLessThan(0.9); + }); + + test('one-sided Fisher exact matches a known table and is one-sided', () => { + expect(fisherOneSidedLower(4, 6, 6, 6)).toBeCloseTo(0.227272727, 8); + expect(fisherOneSidedLower(6, 6, 4, 6)).toBe(1); + expect(fisherOneSidedLower(0, 10, 10, 10)).toBeLessThan(1e-4); + }); + + test('Holm rejects step-down and stops at the first non-rejection', () => { + expect([...holmRejections([0.001, 0.02, 0.04], 0.05)].sort()).toEqual([0, 1, 2]); + expect([...holmRejections([0.001, 0.03, 0.04], 0.05)]).toEqual([0]); + expect([...holmRejections([0.03, 0.04], 0.05)]).toEqual([]); + expect([...holmRejections([], 0.05)]).toEqual([]); + }); +}); + +describe('pass-rates labels', () => { + test('thin history is INCONCLUSIVE, and after this PR every series starts there', () => { + const report = analyze(many('rule-a', 9, 0)); + expect(report.cases[0]).toMatchObject({ case: 'rule-a', label: 'INCONCLUSIVE' }); + expect(formatPassRates(analyze([]))).toContain('every series starts INCONCLUSIVE'); + }); + + test('PASSING, FLAKY and FAILING come from the interval against the entry rate', () => { + expect(analyze(many('rule-a', 12, 0)).cases[0]!.label).toBe('PASSING'); + expect(analyze(many('beh-b', 10, 1)).cases[0]!.label).toBe('FLAKY'); + expect(analyze(many('beh-b', 2, 10)).cases[0]!.label).toBe('FAILING'); + }); + + test('BROKEN: the latest run is 0/n after a prior interval at or above the entry rate', () => { + const prior = many('beh-b', 80, 0, { run_id: 'old' }); + const latest = [1, 2, 3].map(n => trial('beh-b', 'failed', { run_id: 'new', trial: n, panel: { n: 3, k: 2 } })); + expect(analyze([...prior, ...latest]).cases[0]!.label).toBe('BROKEN'); + }); + + test('skipped trials carry no verdict; infra failures count as failed trials', () => { + const stats = analyze([...many('rule-a', 10, 0), trial('rule-a', 'skipped'), + trial('rule-a', 'failed', { failure_class: 'infra' })]).cases[0]!.current!; + expect(stats).toMatchObject({ passes: 10, trials: 11, infra: 1 }); + }); + + test('a new identity, model or CLI starts a new series; earlier series stay visible', () => { + const report = analyze([...many('rule-a', 10, 0), ...many('rule-a', 3, 0, { series_identity: 'id-2' }), + ...many('rule-a', 2, 0, { series_identity: 'id-2', cli_version: '2.1.285' })]); + const c = report.cases[0]!; + expect(c.series).toHaveLength(3); + expect(c.current).toMatchObject({ identity: 'id-2', cli: '2.1.285', trials: 2 }); + expect(c.previous).toMatchObject({ identity: 'id-2', cli: '2.1.284', trials: 3 }); + expect(c.label).toBe('INCONCLUSIVE'); + }); +}); + +describe('pass-rates alarms count post-policy trials of the current series only', () => { + test('backfilled pre-policy failures are displayed but never alarm', () => { + const report = analyze(many('rule-a', 2, 20, { policy_version: 0, source: 'backfill' })); + expect(report.alarms).toEqual([]); + expect(report.cases[0]!.prePolicy).toMatchObject({ passes: 2, trials: 22 }); + expect(report.cases[0]!.label).toBe('INCONCLUSIVE'); + }); + + test('drift proposes quarantine for a blocking case below the entry rule; a rule case is flagged as behaving like behavior', () => { + const kinds = analyze([...many('rule-a', 8, 2), ...many('beh-b', 8, 2), ...many('mar-d', 0, 10)]).alarms.map(a => `${a.kind}:${a.case}`); + expect([...kinds].sort()).toEqual(['drift:beh-b', 'drift:rule-a', 'rule-as-behavior:mar-d', 'rule-as-behavior:rule-a']); + expect(analyze(many('rule-a', 19, 1)).alarms).toEqual([]); + }); + + test('the Fisher regression alarm needs the minimum trials on both sides', () => { + const old = many('gate-c', 6, 0, { series_identity: 'old' }); + const fresh = many('gate-c', 0, 6, { series_identity: 'new' }); + expect(analyze([...old, ...fresh]).alarms.map(a => a.kind)).toContain('regression'); + expect(analyze([...old, ...fresh.slice(0, 5)]).alarms.map(a => a.kind)).not.toContain('regression'); + }); + + test('quarantine exit, expiry and cap', () => { + const exit = analyze(many('beh-b', 10, 0), { 'beh-b': qEntry() }).alarms.map(a => a.kind); + expect(exit).toContain('quarantine-exit'); + expect(exit).not.toContain('drift'); + const weekly = Array.from({ length: 8 }, (_, i) => new Date(Date.UTC(2026, 8, 30) + i * 7 * 86_400_000).toISOString()); + expect(analyze([], { 'beh-b': qEntry() }, { weeklyRuns: weekly }).alarms.map(a => a.kind)).toContain('quarantine-expired'); + expect(analyze([], { 'beh-b': qEntry() }, { weeklyRuns: weekly.slice(0, 7) }).alarms.map(a => a.kind)).not.toContain('quarantine-expired'); + expect(quarantineRunsSince('2026-09-01', undefined, Date.UTC(2026, 9, 27))).toBe(8); + expect(quarantineRunsSince('not a date', undefined, 0)).toBe(Number.POSITIVE_INFINITY); + }); +}); + +describe('quarantine policy', () => { + const policy: PassRatePolicy = EVAL_POLICY; + const now = Date.UTC(2026, 9, 2); + test('a valid entry has no problems', () => { + expect(quarantinePolicyProblems({ 'beh-b': qEntry() }, registry, policy, now)).toEqual([]); + }); + + test('a product defect, a missing diagnosis or field, a bad date or a non-blocking case is rejected', () => { + const problems = (quarantine: Record) => quarantinePolicyProblems(quarantine, registry, policy, now).map(p => p.message); + expect(problems({ 'beh-b': qEntry({ failureClass: 'product' as QuarantineEntry['failureClass'] }) }).join()).toContain('never quarantined'); + expect(problems({ 'beh-b': qEntry({ reason: 'flaky' }) }).join()).toContain('written diagnosis'); + expect(problems({ 'beh-b': qEntry({ owner: ' ' }) }).join()).toContain('missing owner'); + expect(problems({ 'beh-b': qEntry({ enteredAt: '09/29/2026' }) }).join()).toContain('YYYY-MM-DD'); + expect(problems({ 'beh-b': qEntry({ enteredAt: '2027-01-01' }) }).join()).toContain('future'); + expect(problems({ 'mar-d': qEntry() }).join()).toContain('not blocking'); + expect(problems({ 'judge one': qEntry() }).join()).toContain('no registered E2E case'); + expect(problems({ ghost: qEntry() }).join()).toContain('no registered E2E case'); + }); + + test('at most 10% of a tier may be quarantined', () => { + // 11 periodic cases in the fixture registry: the cap is 1. + expect(quarantinePolicyProblems({ 'beh-b': qEntry() }, registry, policy, now)).toEqual([]); + const over = quarantinePolicyProblems({ 'beh-b': qEntry(), 'rule-a': qEntry() }, registry, policy, now); + expect(over.map(p => p.kind)).toEqual(['quarantine-cap']); + expect(over[0]!.message).toContain('2 quarantined periodic cases exceed the 10% cap (1 of 11)'); + }); +}); + +describe('pass-rates inputs', () => { + test('trial-outcomes JSONL is schema-validated; invalid lines are reported, never guessed', () => { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'passrates-')); + const valid = trial('rule-a', 'passed'); + fs.mkdirSync(path.join(dir, 'nested')); + fs.writeFileSync(path.join(dir, 'nested', 'trial-outcomes.jsonl'), formatTrialOutcomes([valid]) + '{"schema":"other"}\nnot json\n'); + fs.writeFileSync(path.join(dir, 'unrelated.jsonl'), formatTrialOutcomes([valid])); + const read = readTrialOutcomeDir(dir); + expect(read.records).toHaveLength(1); + expect(read.records[0]).toMatchObject({ case: 'rule-a', series_identity: 'id-1' }); + expect(read.errors).toHaveLength(2); + fs.rmSync(dir, { recursive: true, force: true }); + }); + + test('legacy records attribute by shard suffix, id, label or single-owner file, else stay unattributed', () => { + expect(attributeLegacyRecord('/anything', 'skill-e2e-b--beh-b', registry)).toBe('beh-b'); + expect(attributeLegacyRecord('rule-a', 'skill-e2e-zzz', registry)).toBe('rule-a'); + expect(attributeLegacyRecord('/gate c labeled', undefined, registry)).toBe('gate-c'); + expect(attributeLegacyRecord('/a display name', 'skill-e2e-a', registry)).toBe('rule-a'); + expect(attributeLegacyRecord('/shared display', 'skill-e2e-shared', registry)).toBeNull(); + }); + + test('backfill keeps only first attempts, defaults a missing attempt to 1, and labels records pre-policy', () => { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'passrates-backfill-')); + fs.writeFileSync(path.join(dir, 'run.json'), run([ + { ...entry_('rule-a', false, 1), exit_reason: 'timeout' }, entry_('rule-a', true, 2), + { name: 'beh-b', suite: 's', tier: 'e2e', passed: true, duration_ms: 1, cost_usd: 0 }, + entry_('/unknown display', true, 1), + ], { shard: 'skill-e2e-zzz' })); + const { records, unattributed } = backfillEvalFiles(collectEvalFiles(dir), { run_id: '42', sha: 'abc' }, registry); + expect(records.map(r => [r.case, r.outcome, r.failure_class, r.policy_version, r.source, r.run_id])) + .toEqual([['rule-a', 'failed', 'timeout', 0, 'backfill', '42'], ['beh-b', 'passed', undefined, 0, 'backfill', '42']]); + expect(formatTrialOutcomes(records)).toContain(TRIAL_OUTCOME_SCHEMA); + expect(unattributed).toEqual(['/unknown display']); + fs.rmSync(dir, { recursive: true, force: true }); + }); + + test('series identity follows the case\'s own touchfiles, not GLOBAL_TOUCHFILES', () => { + const root = fs.mkdtempSync(path.join(os.tmpdir(), 'passrates-series-')); + const git = (...args: string[]) => spawnSync('git', args, { cwd: root, encoding: 'utf8', timeout: 10_000 }); + for (const [file, body] of [['a/x.ts', '1'], ['b/y.ts', '1'], ['harness/run.ts', '1'], ['test/skill-e2e-a.test.ts', '1']] as const) { + fs.mkdirSync(path.join(root, path.dirname(file)), { recursive: true }); + fs.writeFileSync(path.join(root, file), body); + } + const snapshot = () => { expect(git('add', '-A').status).toBe(0); return caseSeriesIdentities(['rule-a', 'beh-b'], root, registry); }; + expect(git('init', '-q').status).toBe(0); + const first = snapshot(); + expect(first['rule-a']).not.toBe(first['beh-b']); + fs.writeFileSync(path.join(root, 'harness/run.ts'), '2'); + expect(snapshot()).toEqual(first); + fs.writeFileSync(path.join(root, 'a/x.ts'), '2'); + const next = snapshot(); + expect(next['rule-a']).not.toBe(first['rule-a']); + expect(next['beh-b']).toBe(first['beh-b']); + fs.rmSync(root, { recursive: true, force: true }); + }); +}); + +describe('pass-rates history fetch (injected, no network)', () => { + function storedZip(files: Record): Buffer { + const locals: Buffer[] = [], centrals: Buffer[] = []; + let offset = 0; + for (const [name, text] of Object.entries(files)) { + const data = Buffer.from(text), fileName = Buffer.from(name), crc = Bun.hash.crc32(data) >>> 0; + const local = Buffer.alloc(30); local.writeUInt32LE(0x04034b50, 0); local.writeUInt16LE(20, 4); + local.writeUInt32LE(crc, 14); local.writeUInt32LE(data.length, 18); local.writeUInt32LE(data.length, 22); local.writeUInt16LE(fileName.length, 26); + const central = Buffer.alloc(46); central.writeUInt32LE(0x02014b50, 0); central.writeUInt16LE(20, 4); central.writeUInt16LE(20, 6); + central.writeUInt32LE(crc, 16); central.writeUInt32LE(data.length, 20); central.writeUInt32LE(data.length, 24); + central.writeUInt16LE(fileName.length, 28); central.writeUInt32LE(offset, 42); + locals.push(local, fileName, data); centrals.push(central, fileName); + offset += 30 + fileName.length + data.length; + } + const size = centrals.reduce((sum, b) => sum + b.length, 0); + const end = Buffer.alloc(22); end.writeUInt32LE(0x06054b50, 0); end.writeUInt16LE(Object.keys(files).length, 8); + end.writeUInt16LE(Object.keys(files).length, 10); end.writeUInt32LE(size, 12); end.writeUInt32LE(offset, 16); + return Buffer.concat([...locals, ...centrals, end]); + } + + test('lists runs per branch, deduplicated and newest first', () => { + const fetcher: HistoryFetcher = { + listRuns: (_repo, _workflow, branch) => branch === 'main' + ? [{ id: 1, attempt: 1, sha: 'a', branch, createdAt: '2026-09-01T00:00:00Z' }, { id: 3, attempt: 1, sha: 'c', branch, createdAt: '2026-09-15T00:00:00Z' }] + : [{ id: 3, attempt: 1, sha: 'c', branch, createdAt: '2026-09-15T00:00:00Z' }, { id: 2, attempt: 2, sha: 'b', branch, createdAt: '2026-09-08T00:00:00Z' }], + listArtifacts: () => [], downloadZip: () => { throw new Error('unused'); }, + }; + expect(listWeeklyRuns({ repo: 'o/r', workflow: 'evals-periodic.yml', branches: ['feature', 'main'], limit: 10, fetcher }).map(r => r.id)).toEqual([3, 2, 1]); + }); + + test('downloads only matching, bounded artifacts once, and caches them', () => { + const cacheDir = fs.mkdtempSync(path.join(os.tmpdir(), 'passrates-cache-')); + const downloads: number[] = []; + const fetcher: HistoryFetcher = { + listRuns: () => [], + listArtifacts: () => [{ id: 10, name: 'trial-outcomes-gate', size: 100 }, { id: 11, name: 'paid-slice-1', size: 100 }, + { id: 12, name: 'trial-outcomes-huge', size: 10 ** 9 }, { id: 13, name: 'trial-outcomes/../escape', size: 1 }], + downloadZip: (_repo, id, destination) => { downloads.push(id); fs.writeFileSync(destination, storedZip({ 'trial-outcomes.jsonl': formatTrialOutcomes([trial('rule-a', 'passed')]) })); }, + }; + const options = { repo: 'o/r', run: { id: 7, attempt: 1, sha: 's', branch: 'main', createdAt: '' }, cacheDir, fetcher, + match: (name: string) => name.startsWith('trial-outcomes') }; + const dirs = downloadRunArtifacts(options); + expect(downloads).toEqual([10]); + expect(dirs).toHaveLength(1); + expect(readTrialOutcomeDir(dirs[0]!).records.map(r => r.case)).toEqual(['rule-a']); + expect(downloadRunArtifacts(options)).toEqual(dirs); + expect(downloads).toEqual([10]); + fs.rmSync(cacheDir, { recursive: true, force: true }); + }); +}); + +describe('pass-rates CLI', () => { + const cli = (args: string[]) => spawnSync(process.execPath, [path.resolve(import.meta.dir, '../scripts/eval-flake-rank.ts'), ...args], + { encoding: 'utf8', timeout: 20_000 }); + + test('--dir prints per-case pass rates; --gate fails only on ACTION REQUIRED', () => { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'passrates-cli-')); + const id = 'plan-ceo-review-format-mode'; + const records = Array.from({ length: 12 }, (_, i) => ({ ...trial(id, i < 11 ? 'passed' : 'failed'), kind: 'behavior' as const, tier: 'periodic' })); + fs.writeFileSync(path.join(dir, 'trial-outcomes.jsonl'), formatTrialOutcomes(records)); + const shown = cli(['--dir', dir, '--case', id]); + expect(shown.status, shown.stderr).toBe(0); + expect(shown.stdout).toContain(`11/12 [`); + expect(shown.stdout).toMatch(new RegExp(`FLAKY\\s+behavior\\s+periodic.*${id}`)); + expect(shown.stdout).toContain('ACTION REQUIRED'); + expect(shown.stdout).toContain(`[drift] ${id} passes 11/12`); + expect(cli(['--dir', dir, '--gate']).status).toBe(1); + fs.writeFileSync(path.join(dir, 'trial-outcomes.jsonl'), formatTrialOutcomes(records.slice(0, 11))); + const clean = cli(['--dir', dir, '--gate', '--json']); + expect(clean.status, clean.stdout).toBe(0); + expect(JSON.parse(clean.stdout).cases[0]).toMatchObject({ case: id, label: 'PASSING', current: { passes: 11, trials: 11 } }); + fs.rmSync(dir, { recursive: true, force: true }); + }); +}); + +function entry_(name: string, passed: boolean, attempt: number) { + return { name, suite: 's', tier: 'e2e', passed, attempt, duration_ms: 1000, cost_usd: 0.1 }; +} diff --git a/test/helpers/periodic-exclude-data.ts b/test/helpers/periodic-exclude-data.ts index 5628e13d7..638c66d18 100644 --- a/test/helpers/periodic-exclude-data.ts +++ b/test/helpers/periodic-exclude-data.ts @@ -106,14 +106,18 @@ export const EVAL_POLICY = { * failures (a product defect is never quarantined), and unchanged case * touchfiles in the change that adds it. Pinned by * test/periodic-exclude-policy.test.ts. - * reason - the written diagnosis - * tracking - issue or TODOS pointer - * owner - who removes it - * enteredAt - ISO date the entry landed (expiry counts weekly runs from here) - * exit - the measurable exit condition + * reason - the written diagnosis, with the pass-rate evidence + * failureClass - what the diagnosis found; a product defect has no class here + * tracking - issue or TODOS pointer + * owner - who removes it + * enteredAt - YYYY-MM-DD the entry landed (expiry counts weekly runs from here) + * exit - the measurable exit condition + * At most EVAL_POLICY.quarantine.capFraction of a tier's cases (gate and + * periodic are the blocking tiers) may be quarantined at once. */ export const CASE_QUARANTINE: Record { } }); }); + +describe('eval verdict policy (pre-registered)', () => { + test('EVAL_POLICY carries exactly the approved constants; a change needs re-approval and a version bump', () => { + expect(EVAL_POLICY).toEqual({ + version: 1, + panel: { n: 3, k: 2 }, + quarantine: { entry: { rate: 0.95, minTrials: 10 }, exit: { rate: 0.97, minTrials: 10 }, capFraction: 0.10, expiryWeeklyRuns: 8 }, + judge: { samples: 3 }, + drift: { fisherAlpha: 0.05, fisherMinPerSide: 6 }, + infraRedispatch: 1, + }); + }); + + test('every CASE_QUARANTINE entry is a diagnosed, dated, non-product blocking case within the tier cap', () => { + expect(quarantinePolicyProblems(CASE_QUARANTINE).map(problem => problem.message)).toEqual([]); + }); + + test('a quarantined case runs as isolated trial shards: its files register it literally', () => { + for (const id of Object.keys(CASE_QUARANTINE)) { + const files = (E2E_TOUCHFILES[id] ?? []).filter(file => /^test\/[^/]+\.test\.ts$/.test(file) && isPaidTestFile(file)); + expect(files.length, `${id}: no paid file registers it`).toBeGreaterThan(0); + for (const file of files) { + expect(fileCaseRegistration(file, fs.readFileSync(path.join(ROOT, file), 'utf8')).known, `${id}: ${file} registration must be statically known`).toBe(true); + } + } + }); +}); From 8ee9f887ed19cbcb542905261936d67532528bd9 Mon Sep 17 00:00:00 2001 From: garrytan Date: Tue, 29 Sep 2026 19:13:54 +0000 Subject: [PATCH 08/15] feat(eval-pass-rates): attribute legacy records by the exact slug of their display name --- scripts/eval-flake-rank.ts | 7 +++++-- test/eval-flake-rank.test.ts | 2 ++ 2 files changed, 7 insertions(+), 2 deletions(-) diff --git a/scripts/eval-flake-rank.ts b/scripts/eval-flake-rank.ts index ca46a3eb8..363fd66d8 100644 --- a/scripts/eval-flake-rank.ts +++ b/scripts/eval-flake-rank.ts @@ -189,14 +189,17 @@ export function caseTier(id: string, registry: Registry = LIVE_REGISTRY): string /** * Attribute a legacy eval-store record to a registry id: the case-shard slug - * suffix (`--`), the recorded name, a CASE_TEST_NAMES label, or the - * only id its shard file registers. Anything else is unattributed (null). + * suffix (`--`), the recorded name or its exact slug (`/qa b6-static` + * is `qa-b6-static`), a CASE_TEST_NAMES label, or the only id its shard file + * registers. Anything else is unattributed (null). */ export function attributeLegacyRecord(name: string, shard: string | undefined, registry: Registry = LIVE_REGISTRY): string | null { const known = (id: string) => id in registry.kinds; const [slugFile, slugCase] = (shard ?? '').split('--'); if (slugCase && known(slugCase)) return slugCase; if (known(name)) return name; + const slug = name.toLowerCase().replace(/[^a-z0-9]+/g, '-').replace(/^-+|-+$/g, ''); + if (known(slug)) return slug; const labeled = Object.entries(registry.testNames).find(([, label]) => label === name)?.[0]; if (labeled && known(labeled)) return labeled; if (slugFile) { diff --git a/test/eval-flake-rank.test.ts b/test/eval-flake-rank.test.ts index 0eaa1ddab..f3d815c72 100644 --- a/test/eval-flake-rank.test.ts +++ b/test/eval-flake-rank.test.ts @@ -281,6 +281,8 @@ describe('pass-rates inputs', () => { test('legacy records attribute by shard suffix, id, label or single-owner file, else stay unattributed', () => { expect(attributeLegacyRecord('/anything', 'skill-e2e-b--beh-b', registry)).toBe('beh-b'); expect(attributeLegacyRecord('rule-a', 'skill-e2e-zzz', registry)).toBe('rule-a'); + expect(attributeLegacyRecord('/Rule a', 'skill-e2e-zzz', registry)).toBe('rule-a'); + expect(attributeLegacyRecord('/rule a extra', 'skill-e2e-zzz', registry)).toBeNull(); expect(attributeLegacyRecord('/gate c labeled', undefined, registry)).toBe('gate-c'); expect(attributeLegacyRecord('/a display name', 'skill-e2e-a', registry)).toBe('rule-a'); expect(attributeLegacyRecord('/shared display', 'skill-e2e-shared', registry)).toBeNull(); From ccb5f3c07c30316726e954252bd8e42b22234f66 Mon Sep 17 00:00:00 2001 From: garrytan Date: Tue, 29 Sep 2026 19:17:12 +0000 Subject: [PATCH 09/15] docs(evals): document the pre-registered verdict policy, quarantine, pass-rate history and arithmetic AGENTS.md replaces the retry rule with the approved policy text (no retries; kind fixes trials; no added trials, samples or dispatches after a result; quarantine by CASE_QUARANTINE only; one INFRA/INCOMPLETE re-dispatch) and notes that a pre-registered fixed panel is not rejudging. CONTRIBUTING gains the kind rules, the judge panel, eval:pass-rates and an 'Add a paid eval' checklist. TESTING_INTERNALS describes verdicts, quarantine, history and the arithmetic, including the rule term: 1 trial vs 2-of-3 red rates at p = 0.99/0.95/0.90/0.70/0.30 and lane all-green probabilities for the current 191 rule / 22 behavior / 25 judge registry. --- AGENTS.md | 18 +++-- CONTRIBUTING.md | 51 ++++++++++++-- docs/TESTING_INTERNALS.md | 140 +++++++++++++++++++++++++++++++++----- 3 files changed, 182 insertions(+), 27 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index a68337ffd..c847ad649 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -148,7 +148,8 @@ When fixing failures or preparing `/ship`, follow this order: public events in free regressions, including negative controls, before paying for another agent run. Check behavior and acknowledgments; match exact prose only when that prose is the contract. Do not lower thresholds, increase model - budgets, skip cases, or rejudge a failure to manufacture a pass. + budgets, skip cases, or rejudge a failure to manufacture a pass. A + pre-registered fixed panel is not rejudging. For policy or validation repairs, exercise the actual registered callback with representative native input and assert that it uses the helper’s result. When renderer or parser failures recur at the same boundary, verify the @@ -208,10 +209,16 @@ When fixing failures or preparing `/ship`, follow this order: result and pending permission state; diagnose a blocked actor before waiting through its deadline. Preserve cancellation separately from a test verdict. Skipped or unstarted cases - do not satisfy coverage; preserve every attempt. Retries follow the approved - policy in `test/helpers/eval-budgets.ts`: a timed-out attempt is a verdict, so - only files whose every case budget is CAPTURE tier or shorter keep one retry; - never add retries to pass a longer case. + do not satisfy coverage; preserve every attempt. Paid evals never retry. Each + case's kind (`E2E_KINDS`) fixes its trials before the run: `rule` one trial; + `behavior` a panel of 3 independent trials, PASS at >= 2 with no contract + violation; `judge` 3 samples on one output, gated on the mean against the + unchanged threshold. Never add trials, samples or dispatches after seeing a + result, never change a kind to change a verdict without pass-rate evidence, + and report every trial. Quarantine follows `CASE_QUARANTINE`'s entry and exit + rules only (`EVAL_POLICY`, `docs/TESTING_INTERNALS.md`). A census whose every + red is machine-classified INFRA or INCOMPLETE may be re-dispatched once as a + new run; report both runs. 7. Prove all known repairs with focused tests, including affected paid cases. Rerun a failed case only after a concrete repair or a demonstrated launch correction. Run the remaining required selected evaluations on the integrated @@ -245,6 +252,7 @@ bun run test # complete free suite via the strict shard runner (no A bun run test:ubicloud # same suite on an ephemeral 16-vCPU Ubicloud VM (needs UBICLOUD_API_KEY) bun run eval:bg:pr # changed fast live probes + selected judges, with explicit deferrals bun run eval:bg:release # fresh complete gate + periodic live coverage +bun run eval:pass-rates # per-case trial pass rates (Wilson), drift and quarantine alarms (--case, --gate) bun run scripts/test-paid-shards.ts --tier periodic --list --slice-budget 540 --jobs 2 # CI slice plan preview (free) bun run test:windows # curated Windows-safe subset (runs on windows-latest) bun run build # generate docs + compile binaries diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index db9f74c8e..07496edbb 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -239,10 +239,29 @@ Complete start-to-finish flows belong to the `marathon` tier (`describeE2ETier('marathon')`), which runs only in the non-blocking `evals-marathon.yml` lane (weekly and on dispatch) and never gates a merge. -Retries: a timed-out attempt is a verdict. Only files whose every case budget is -CAPTURE tier or shorter (`RETRY_MAX_CASE_MS` in `test/helpers/eval-budgets.ts`) -keep one automatic retry for fast-failing flakes; every other paid file runs once. -Case budgets themselves never change with this rule. +Verdicts: paid evals never retry. Each case's kind in `E2E_KINDS` +(`test/helpers/touchfiles-data.ts`) fixes its trials before the run, from the +constants in `EVAL_POLICY` (`test/helpers/periodic-exclude-data.ts`): + +- `rule` (the default): one trial; any failed assertion fails the case. Use it + when nothing stochastic decides the verdict, or when the verdict checks a + contract the product must meet every run (no writes in plan mode, a question + before a decision, a skill-mandated step, no leaked secret). +- `behavior`: a panel of 3 independent trials run as parallel case shards, + PASS at 2 or more with no contract violation (`expectContract()`). Use it only + when a live model choice decides the verdict and an occasional deviation is + acceptable product behavior; the one-line reason goes in `BEHAVIOR_WHY`. +- `judge`: an LLM judge scoring a fixed input; 3 samples of the same prompt, + gated on the per-dimension mean (booleans on a majority) against the + unchanged threshold. An erroring sample fails the panel and is never resampled. + +A timed-out, crashed or infrastructure-failed trial counts as a failed trial and +is reported with its class; a missing trial makes the case INCOMPLETE, which +fails the lane. A 2-of-3 pass is reported as `PASS 2/3` with the failed trial's +cause, never as a clean pass. Case budgets and thresholds never change with +this policy. Quarantine (`CASE_QUARANTINE`) and history are described in +`docs/TESTING_INTERNALS.md`; `bun run eval:pass-rates --case ` shows a +case's per-trial pass rate with its Wilson interval. CI enables verified first-attempt reuse for 16 workflow quality judges for 24 hours within the same PR. The cookie workflow's custom input and the other 11 @@ -415,7 +434,7 @@ When E2E tests run, they produce machine-readable artifacts in `~/.gstack-dev/`: bun run eval:list # list all eval runs (turns, duration, cost per run) bun run eval:compare # compare two runs — shows per-test deltas + Takeaway commentary bun run eval:summary # aggregate stats + per-test efficiency averages across runs -bun run eval:flake-rank # rank tests by flake signal: retried passes first, then failure rate (--json, --dir, --since-days) +bun run eval:pass-rates # per-case trial pass rates + Wilson intervals from recent weekly runs (--case, --runs, --dir, --backfill, --json, --gate); eval:flake-rank is an alias ``` **Detached runs for agents and long suites.** When an agent (or you, for a run @@ -463,7 +482,9 @@ Override the judge model per run with `GSTACK_EVAL_MODEL_JUDGE`: - **Completeness** — Are all commands, flags, and usage patterns documented? - **Actionability** — Can the agent execute tasks using only the information in the doc? -Each dimension is scored 1-5. Threshold: every dimension must score **≥ 4**. There's also a regression test that compares generated docs against the hand-maintained baseline from `origin/main` — generated must score equal or higher. +Each dimension is scored 1-5 by a panel of 3 samples of the same prompt, drawn +concurrently; each dimension's panel mean must meet that judge's threshold (≥ 4 +for most dimensions; see each case). An erroring sample fails the panel. There's also a regression test that compares generated docs against the hand-maintained baseline from `origin/main` — generated must score equal or higher. ```bash # Needs ANTHROPIC_API_KEY in .env — included in bun run test:evals @@ -483,6 +504,24 @@ fails, add the named path to the named key and check selection with `bun run scripts/test-paid-shards.ts --tier gate --profile pr --list`. The rule is a lower bound: a fixture path the test builds at runtime is not visible to it, so add such paths to the key by hand. +### Add a paid eval + +1. **Test file.** Write the case in a paid test file, registered with a literal + name (`testIfSelected('', ...)`), grading the outcome (files, git + state, native questions, exit status) rather than wording, unless the step + itself is the contract. Wrap contract assertions in `expectContract()`. +2. **Touchfiles.** Add `'': [...]` to `E2E_TOUCHFILES`; `bun test + test/touchfiles.test.ts` names any missing closure path. +3. **Tier.** Add it to `E2E_TIERS`: `gate` for cheap contracts every PR needs, + `periodic` for long or model-quality cases, `marathon` for complete flows. +4. **Kind.** Add it to `E2E_KINDS` (`rule` unless a live model choice may + acceptably deviate; then `behavior` plus a `BEHAVIOR_WHY` line). + `bun test test/eval-kinds.test.ts` prints the literal to add. +5. **PR profile.** If a PR should run it, add it to `scripts/test-pr-profile.ts` + and check `bun run scripts/test-paid-shards.ts --tier gate --profile pr --list`. +6. **Try the panel locally.** `bun run scripts/test-paid-shards.ts --tier + --case --trials 3` runs the same panel CI runs, before you push. + ### CI A GitHub Action (`.github/workflows/skill-docs.yml`) generates all hosts on pushes to main and on PRs, then rejects tracked differences and nonignored untracked output. Generation errors also fail the job. Optional ignored host caches are not compared against Git. diff --git a/docs/TESTING_INTERNALS.md b/docs/TESTING_INTERNALS.md index 36d89ff3c..0cb64ac28 100644 --- a/docs/TESTING_INTERNALS.md +++ b/docs/TESTING_INTERNALS.md @@ -252,24 +252,20 @@ processes × `EVALS_CONCURRENCY` within-shard, per-shard `GSTACK_EVAL_DIR`, full-stream spooling to per-shard log files (path printed at START and on failure), never-started/timed-out taxonomy, and parent-computed diff selection propagated to children via `EVALS_SELECTION_JSON` (fail-open: a -child that can't parse it recomputes locally with one warning). Retries follow -one rule (`retriesForFiles`, `RETRY_MAX_CASE_MS` in `test/helpers/eval-budgets.ts`): -a timed-out attempt is a verdict, so a file keeps one Bun retry only when every -case budget is CAPTURE tier plus its recording grace or shorter (registered rows -derive it from `caseMs`, `SHORT_CASE_RETRY_FILES` lists the rest); every other -file, including the former `retries: 2` matrix rows, runs once. `--list` prints -each shard's retries. Files in `CASE_SHARDED_FILES` run one registered case per +child that can't parse it recomputes locally with one warning). Paid evals +never retry; each case's kind fixes its trials before the run (see "Eval verdict +policy" below). Files in `CASE_SHARDED_FILES` run one registered case per process (`#`, an exact `--test-name-pattern`, exactly one executed case), so a long file of short cases spreads across runners and each case gets its own SDK semaphore. -Flake telemetry rides the store: every recorded test carries its 1-based -`attempt` (a pass-on-attempt-2 stays visible forever — bun's own stream hides -it), runs list `flaky_retries`, the report warns on passed-only-on-retry -tests, and `bun run eval:flake-rank` ranks the series (retried passes first, -then failure rate; 60-day recency bound on eval files; the free lane's flake -ledger is folded in from `flakeLedgerPath()` — override with -`GSTACK_FLAKE_LEDGER`, the same env var the CI free lane sets before -uploading the ledger as the `flake-ledger` artifact). Census integrity is +Trial telemetry rides the store: every recorded test carries its 1-based +`attempt` plus, on an isolated trial shard, its `case_id`, `kind`, `trial`, +`panel` and `policy_version`, and each lane's report uploads one +`trial-outcomes` JSONL line per trial. `bun run eval:pass-rates` +(`eval:flake-rank` is an alias) turns that history into per-case pass rates +(see "Pass-rate history" below; the free lane's flake ledger is folded in from +`flakeLedgerPath()` — override with `GSTACK_FLAKE_LEDGER`, the same env var the +CI free lane sets before uploading the ledger as the `flake-ledger` artifact). Census integrity is enforced from the free suite: every `E2E_TOUCHFILES` / `LLM_JUDGE_TOUCHFILES` key must name a living paid test (`test/touchfiles.test.ts`'s reverse invariant), and `git show :path` fixtures are banned — vendor the bytes @@ -313,7 +309,7 @@ CI supplies the scoped cache/runtime configuration; local runs are fresh by default. Cached scores must pass current assertions; reused records retain their original source and time and cannot renew the receipt. `scripts/e2e-shard-reuse.ts` extends the same receipts -to PR-profile E2E shards that run with zero retries (so the pass is structurally a +to PR-profile E2E shards (paid evals never retry, so a pass is structurally a first attempt): the identity hashes the test's import closure, every tracked file matched by the touchfiles of every case the file registers plus the global touchfiles, the runner/workflow/setup actions, the child's environment pins, the @@ -377,6 +373,118 @@ the runner parent and handed to shard children as `GSTACK_CLAUDE_CLI_VERSION` (never spawned on a test thread), so a TUI-drift flake hunt is a grep, not archaeology. +**Eval verdict policy** (`EVAL_POLICY` version 1 in +`test/helpers/periodic-exclude-data.ts`, pre-registered 2026-09-29). Paid evals +never retry. Each live case has exactly one kind in `E2E_KINDS` +(`test/helpers/touchfiles-data.ts`; `test/eval-kinds.test.ts` enforces coverage), +and the kind fixes its trials before the run: + +- `rule` (default): one trial; any failed assertion fails the verdict. For + cases where nothing stochastic decides the verdict, or where it checks a + contract the product must meet every run. +- `behavior`: a panel of `n = 3` independent trials, launched together as + isolated case shards on different slices (key `#~t`). All three + always run: no early stop and no conditional extra trial. PASS when at least + `k = 2` pass and no trial violated a contract (`expectContract()` stamps + `failure_class: 'contract'`). Each behavior case names its tolerated deviation + in `BEHAVIOR_WHY` and must have a literal registration so it can run alone. +- `judge`: an LLM judge scoring a fixed input. `judgePanel()` + (`test/helpers/llm-judge.ts`) draws 3 samples of the same prompt concurrently + inside the unchanged `JUDGE_MS`; numeric dimensions gate on the per-dimension + mean against the unchanged threshold (no dimension compensates for another), + booleans on a strict majority. A sample that errors (refusal, truncation, + non-JSON, a malformed field) fails the panel and is never resampled; a + refusal counts as an unscored panel only when every sample refused. + `callJudge`'s 429 backoff happens before any model output and is transport, + not a verdict retry. The workflow-judge cache stores whole panels only. + +`panelVerdict()` (`test/helpers/eval-store.ts`) is the single verdict +function the report, `collector-outcomes.json`, the PR comment and pass-rates +all use. A timed-out, crashed or infrastructure-failed trial is a failed trial +recorded with its class; a missing or duplicate trial record makes the verdict +INCOMPLETE, which fails the lane; a 2/3 PASS is shown as `PASS 2/3` with the +failed trial's cause. A manual re-run adds trials under a new run attempt and +never replaces the first attempt's verdict. A red census is never rerun on +unchanged inputs: each red is diagnosed as product, test/detector, harness or +infra and resolved by a concrete repair and a fresh census, or listed as a named +red. The one exception: a census whose every red verdict is machine-classified +INFRA or INCOMPLETE (missing slice artifact, runner loss, API error before the +first model turn) may be re-dispatched once as a new run, and both runs are +reported. Changing any `EVAL_POLICY` constant after seeing census results needs +Garry's re-approval, a `version` bump and a fresh census; +`test/periodic-exclude-policy.test.ts` pins the approved values. + +**Quarantine** (`CASE_QUARANTINE`, same file). An entry needs: a per-trial rate +below 95% over at least 10 post-policy trials of the case's current input +identity (pre-policy backfill may justify only an initial entry, labeled as +such); a written diagnosis in `reason` whose `failureClass` is `detector`, +`harness` or `model-latency` (a product defect is fixed or listed as a named +red, never quarantined); unchanged case touchfiles in the change that adds it; +and an owner, tracking pointer, `enteredAt` date and measurable `exit`. A +quarantined case still runs its full panel and reports in every lane but cannot +fail it, except on a hard break (0 of n) or a contract violation, and it never +counts as passing coverage. At most 10% of a blocking tier (gate, periodic) may +be quarantined. The weekly report fails when an entry passes its exit rule (at +least 97% over at least 10 trials) without being removed, when an entry is 8 +weekly runs old, or when a tier is over its cap. + +**Pass-rate history** (`bun run eval:pass-rates`, `scripts/eval-flake-rank.ts`). +It reads the `trial-outcomes` artifact of the last N completed +`evals-periodic.yml` runs on the current branch and `main` (flags: `--case`, +`--runs N`, `--branch`, `--dir`, `--backfill`, `--json`, `--gate`) and prints +per-case per-trial pass rates with 95% Wilson intervals. A series is one case +under one input identity, the hash of its own touchfiles minus +`GLOBAL_TOUCHFILES` (harness edits do not restart it), per model, Claude CLI +version and policy version; a change starts a new series and older ones stay +visible. Labels: INCONCLUSIVE below 10 trials, BROKEN when the latest run is +0/n after a prior interval at or above 95%, FLAKY when failures leave the +interval straddling 95%, FAILING when the whole interval is below it, PASSING +otherwise. `--backfill` imports legacy slice artifacts as pre-policy trials +(first attempt only; a record that names no registry id is listed as +unattributed, never guessed); they are display-only. `--gate` (the weekly +report) fails with ACTION REQUIRED, on post-policy trials of the current series +only, when a non-quarantined blocking case meets the entry rule (proposing an +entry), when a `rule` case does (rule case behaving like behavior: fix or +reclassify), when a blocking case's current identity is significantly below its +previous one (one-sided Fisher exact, α = 0.05, at least 6 trials each side, +Holm-controlled across the cases tested), and on the quarantine rules above. +History that cannot be fetched fails the gate closed. + +**The arithmetic.** With per-trial pass rate p, the chance a single case goes +red (a false red while the product works, the catch rate once it has +regressed): + +| p | 1 trial | 2-of-3 panel | +|---|---|---| +| 0.99 | 1.0% | 0.03% | +| 0.95 | 5.0% | 0.72% | +| 0.90 | 10.0% | 2.8% | +| 0.70 | 30.0% | 21.6% | +| 0.30 | 70.0% | 78.4% | + +The panel removes most false reds at healthy rates, but it catches a 0.95 → 0.70 +regression in one run only 21.6% of the time (a single trial 30%, retry-until-green +3%), so drift detection is the history rule's job, not the per-run verdict's. +The Fisher alarm is weak at the minimum sample (5.4% power for 0.95 → 0.70 at +6 trials a side), and ten straight passes still leave a 72% Wilson lower bound: +after this policy lands, every series starts INCONCLUSIVE. + +A lane is all green with probability Π p_rule × Π P(≥2 of 3 | p_behavior) × +Π p_judge. For the current registry (PR gate worst case: 107 rule cases and 24 +judges; weekly census: 190 rule, 22 behavior and 25 judge verdicts), with rule +and judge verdicts at p_rule: + +| p_rule | full PR gate | weekly, behavior p = 0.90 | 0.95 | 0.97 | +|---|---|---|---|---| +| 0.99 | 26.8% | 6.2% | 9.8% | 10.9% | +| 0.995 | 51.9% | 18.2% | 29.0% | 32.1% | +| 0.999 | 87.7% | 43.2% | 68.7% | 76.1% | + +The rule term dominates: a green lane on a working product needs rule cases to +be near-deterministic (0.999), which is why failing detectors are converted to +outcome checks and product defects are fixed or named, and why each census +reports its expected lane false-red from the measured rates. + **Timeout policy.** Paid tests use the tiers in `test/helpers/eval-budgets.ts` (JUDGE/CAPTURE/CAPTURE_LONG/PTY/PTY_LONG); `test/eval-budgets-policy.test.ts` pins that every tier fits the shard wall From 8eb55b6ebf4a23a3e75eff5416e3c86c77e9273c Mon Sep 17 00:00:00 2001 From: garrytan Date: Tue, 29 Sep 2026 19:23:55 +0000 Subject: [PATCH 10/15] feat(evals): trial planner, slice exit split and panel-verdict report Planner: behavior and quarantined cases become panels of isolated trial shards (#~t) bound by EVALS_SELECTION_JSON=[id] and the exact test name; the file shard excludes them by name. Trials of one case never share a slice, result slugs are unique, panels are validated whole, unknown registrations throw, and the planner prints a capacity preflight. Executor: each trial shard gets its TRIAL_ENV identity and a trial record (outcome, failure class, cause, cost); every shard writes a JUnit report. The slice exit now means execution completeness: a failed rule shard or a trial without a record reds the runner, a failed trial does not. Report: panelVerdict() decides every panel of the first run attempt (later attempts are reported, never replacing it); rule shards keep the unchanged fail-closed checks; collector records all count (no last-attempt wins); census runs enforce the quarantine cap and expiry. It writes collector-outcomes v2, trial-outcomes.jsonl (trials plus JUnit rule/judge cases), report-summary.md, and one headline + failure block with rerun commands, and flags INFRA/INCOMPLETE-only reds for the one re-dispatch. The fail-open suite gains the panel cases: behavior 1/3 red, 2/3 green with its failed trial shown, missing trial INCOMPLETE, contract at 2/3 red, quarantined 1/3 green, 0/3 and contract red, missing slice red, and a later attempt never replacing the first. --- scripts/test-paid-shards.ts | 1091 +++++++++++++++++++++++----- test/ci-paid-coordination.test.ts | 3 +- test/paid-pr-profile.test.ts | 5 +- test/paid-report-fail-open.test.ts | 107 +++ test/paid-run-manifest.test.ts | 100 +++ test/paid-shards.test.ts | 97 +++ 6 files changed, 1225 insertions(+), 178 deletions(-) diff --git a/scripts/test-paid-shards.ts b/scripts/test-paid-shards.ts index c4f7031b8..4fdd84b8e 100644 --- a/scripts/test-paid-shards.ts +++ b/scripts/test-paid-shards.ts @@ -66,9 +66,14 @@ import { type ShardChildResult, } from './test-strict-output'; import { PAID_TEST_GLOBS, isPaidTestFile } from '../test/helpers/paid-test-set'; -import { CASE_CI_EXCLUDE, PERIODIC_CI_EXCLUDE } from '../test/helpers/periodic-exclude-data'; +import { CASE_CI_EXCLUDE, CASE_QUARANTINE, EVAL_POLICY, PERIODIC_CI_EXCLUDE } from '../test/helpers/periodic-exclude-data'; import { FILE_RETRY_BUDGETS, STRICT_RETRY_CASE_BUDGETS } from '../test/helpers/eval-budgets'; -import { getProjectEvalDir, getClaudeCliVersion, isFinalizedEvalResultFile, evalEntryOutcome } from '../test/helpers/eval-store'; +import { + getProjectEvalDir, getClaudeCliVersion, isFinalizedEvalResultFile, evalEntryOutcome, failureClassOf, panelVerdict, + sanitizeTrialError, formatTrialOutcomes, CONTRACT_VIOLATIONS_FILE, TRIAL_ENV, TRIAL_OUTCOME_SCHEMA, TRIAL_OUTCOMES_FILE, + type EvalCaseKind, type PanelShape, type PanelVerdict, type TrialFailureClass, type TrialOutcome, type TrialOutcomeRecord, +} from '../test/helpers/eval-store'; +import { E2E_KINDS } from '../test/helpers/touchfiles-data'; import { manualReviewProblem } from '../test/helpers/cookie-workflow-manual-review'; import { preflightAnthropicApi } from '../test/helpers/anthropic-preflight'; import { OVERLAY_MIN_FILE_WALL_MS } from '../test/helpers/overlay-case-policy'; @@ -146,16 +151,60 @@ export const CASE_TEST_NAMES: Record = { }; const CASE_KEY_SEPARATOR = '#'; +const TRIAL_SUFFIX = /~t([1-9][0-9]*)$/; -/** The test file behind a shard key (`` or `#`). */ +/** The test file behind a shard key (``, `#` or `#~t`). */ export function shardFile(key: string): string { return normalizeRelativePath(key).split(CASE_KEY_SEPARATOR)[0]!; } -/** The E2E case id of a case shard key, else null. */ +/** The E2E case id of a case or trial shard key, else null. */ export function shardCaseId(key: string): string | null { const [, id] = normalizeRelativePath(key).split(CASE_KEY_SEPARATOR); - return id ?? null; + return id === undefined ? null : id.replace(TRIAL_SUFFIX, ''); +} + +/** The 1-based trial index of an isolated trial shard key, else null. */ +export function shardTrial(key: string): number | null { + const [, id] = normalizeRelativePath(key).split(CASE_KEY_SEPARATOR); + const match = id === undefined ? null : TRIAL_SUFFIX.exec(id); + return match ? Number(match[1]) : null; +} + +/** Shard key of one trial of an isolated case. */ +export function trialShardKey(file: string, id: string, trial: number): string { + return `${normalizeRelativePath(file)}${CASE_KEY_SEPARATOR}${id}~t${trial}`; +} + +/** Trial policy of one case, fixed from the registries before the run. */ +export interface CaseTrialPlan { kind: EvalCaseKind; panel: PanelShape; quarantined: boolean } + +/** + * `behavior` cases run EVAL_POLICY.panel; a quarantined case runs a full panel + * whose k keeps its kind's meaning (k = n for rule); everything else runs one + * trial. Only behavior and quarantined cases are isolated into trial shards. + */ +export function caseTrialPlan(id: string, kinds: Record = E2E_KINDS, + quarantine: Record = CASE_QUARANTINE): CaseTrialPlan { + const kind = kinds[id] ?? 'rule'; + const quarantined = Object.hasOwn(quarantine, id); + if (kind === 'behavior') return { kind, panel: { ...EVAL_POLICY.panel }, quarantined }; + if (quarantined) return { kind, panel: { n: EVAL_POLICY.panel.n, k: EVAL_POLICY.panel.n }, quarantined }; + return { kind, panel: { n: 1, k: 1 }, quarantined }; +} + +export function isIsolatedCase(plan: CaseTrialPlan): boolean { + return plan.kind === 'behavior' || plan.quarantined; +} + +function sameTrialPlan(a: CaseTrialPlan | undefined, b: CaseTrialPlan | undefined): boolean { + return !!a && !!b && a.kind === b.kind && a.quarantined === b.quarantined && a.panel?.n === b.panel?.n && a.panel?.k === b.panel?.k; +} + +/** Bun name pattern that runs every case of a file except `ids` (their trial shards run them). */ +export function excludedCasesNamePattern(ids: string[]): string { + const escaped = ids.map(id => (CASE_TEST_NAMES[id] ?? id).replace(/[.*+?^${}()|[\]\\]/g, '\\$&')); + return `^(?!.*(?:^|\\s)(?:${escaped.join('|')})$)`; } /** Exact Bun name pattern for a set of case ids (labels where the test name differs). */ @@ -180,6 +229,61 @@ export function expandCaseShards(files: string[], tier: PaidTier, rootDir = ROOT }); } +export interface TrialExpansion { + keys: string[]; + /** Trial policy per trial shard key. */ + trials: Record; + /** File shard key -> isolated case ids its name pattern excludes. */ + excludeCases: Record; +} + +/** + * Isolate every behavior or quarantined case of `tier` into its panel of trial + * shards (`#~t1..tn`), each selected by EVALS_SELECTION_JSON=[id] and + * its exact test name. The file shard keeps the remaining ids of the tier and + * excludes the isolated ones by name; with none remaining it is dropped. A case + * may be isolated only when its file's registration is statically known. + */ +export function expandTrialShards(keys: string[], tier: PaidTier, rootDir = ROOT, opts: { + kinds?: Record; quarantine?: Record; + touchfiles?: Record; tiers?: Record; +} = {}): TrialExpansion { + const touchfiles = opts.touchfiles ?? E2E_TOUCHFILES; + const tiers = opts.tiers ?? E2E_TIERS; + const planOf = (id: string) => caseTrialPlan(id, opts.kinds, opts.quarantine); + const out: TrialExpansion = { keys: [], trials: {}, excludeCases: {} }; + const addPanel = (file: string, id: string) => { + const plan = planOf(id); + for (let trial = 1; trial <= plan.panel.n; trial++) { + const key = trialShardKey(file, id, trial); + out.keys.push(key); + out.trials[key] = plan; + } + }; + for (const key of keys) { + const file = shardFile(key); + const caseId = shardCaseId(key); + if (caseId !== null) { + if (isIsolatedCase(planOf(caseId))) addPanel(file, caseId); + else out.keys.push(key); + continue; + } + const { registered, known } = fileCaseRegistration(file, fs.readFileSync(path.join(rootDir, file), 'utf8'), touchfiles, tiers); + const inTier = registered.filter(id => tiers[id] === tier); + const isolated = inTier.filter(id => isIsolatedCase(planOf(id))); + if (isolated.length === 0) { out.keys.push(key); continue; } + if (!known) { + throw new Error(`${file}: behavior or quarantined case(s) ${isolated.join(', ')} need a statically known case registration`); + } + for (const id of isolated) addPanel(file, id); + if (inTier.length > isolated.length) { + out.keys.push(key); + out.excludeCases[normalizeRelativePath(key)] = [...isolated].sort(); + } + } + return out; +} + /** * Split expanded shard keys into runnable keys and CI-unrunnable cases * (CASE_CI_EXCLUDE), each with its surfaced reason; never an empty shard. @@ -484,26 +588,27 @@ function packageVersionOnlySinceBase(rootDir: string, baseRef: string): boolean /** Only audited per-case files, plus the separately selected judge, enter the fast profile. */ /** The selected PR-profile case ids a shard key owns (a case key owns at most its own case). */ -function prProfileShardIds(key: string, selection: PaidCaseSelection): string[] { +/** `exclude`: isolated case ids a file shard leaves to their trial shards. */ +function prProfileShardIds(key: string, selection: PaidCaseSelection, exclude: readonly string[] = []): string[] { const caseId = shardCaseId(key); return (PR_PROFILE_FILES[shardFile(key)] ?? []) - .filter(id => (caseId === null || id === caseId) && (selection.e2e === null || selection.e2e.includes(id))); + .filter(id => (caseId === null || id === caseId) && (selection.e2e === null || selection.e2e.includes(id)) && !exclude.includes(id)); } -export function prProfileFileSelected(file: string, selection: PaidCaseSelection): boolean { +export function prProfileFileSelected(file: string, selection: PaidCaseSelection, exclude: readonly string[] = []): boolean { if (file === 'test/skill-llm-eval.test.ts') return selection.judges === null || selection.judges.length > 0; - return prProfileShardIds(file, selection).length > 0; + return prProfileShardIds(file, selection, exclude).length > 0; } -export function expectedPrCaseCount(file: string, selection: PaidCaseSelection): number { +export function expectedPrCaseCount(file: string, selection: PaidCaseSelection, exclude: readonly string[] = []): number { if (file === 'test/skill-llm-eval.test.ts') return selection.judges?.length ?? Object.keys(LLM_JUDGE_TOUCHFILES).length; - return prProfileShardIds(file, selection).length; + return prProfileShardIds(file, selection, exclude).length; } -export function prProfileTestNamePattern(file: string, selection: PaidCaseSelection): string { +export function prProfileTestNamePattern(file: string, selection: PaidCaseSelection, exclude: readonly string[] = []): string { const ids = file === 'test/skill-llm-eval.test.ts' ? selection.judges ?? Object.keys(LLM_JUDGE_TOUCHFILES) - : prProfileShardIds(file, selection); + : prProfileShardIds(file, selection, exclude); if (ids.length === 0) throw new Error(`No selected PR cases for ${file}`); return caseTestNamePattern(ids); } @@ -527,6 +632,8 @@ export interface DiffSkipOptions { allNames?: string[]; /** Injectable registration map (default: E2E_TOUCHFILES). */ e2eTouchfiles?: Record; + /** File shard -> isolated case ids its trial shards run instead. */ + excludeCases?: Record; } /** @@ -569,7 +676,8 @@ export function diffSkipDecisionForFile( const touchfiles = options.e2eTouchfiles ?? E2E_TOUCHFILES; const quoted = knownTestNamesInSource(source, allNames); const registered = Object.keys(touchfiles).filter((k) => touchfiles[k].includes(rel)); - const mapped = [...new Set([...quoted, ...registered])]; + const isolated = options.excludeCases?.[rel] ?? []; + const mapped = [...new Set([...quoted, ...registered])].filter(name => !isolated.includes(name)); if (mapped.length === 0) { return { file, kept: true, reason: 'no mappable test names — fail-open, child self-skip authoritative' }; } @@ -679,7 +787,8 @@ export function buildPaidShardArgs( export function shardSlug(files: string[]): string { return files .map((file) => path.basename(shardFile(file)).replace(/\.test\.(?:[cm]?[jt]s|tsx|jsx)$/, '') - + (shardCaseId(file) === null ? '' : `--${shardCaseId(file)}`)) + + (shardCaseId(file) === null ? '' : `--${shardCaseId(file)}`) + + (shardTrial(file) === null ? '' : `.t${shardTrial(file)}`)) .join('+') .replace(/[^a-zA-Z0-9._+-]/g, '-'); } @@ -715,6 +824,91 @@ export interface ShardOutcome { budget?: PaidShardBudget; /** Present when a verified receipt replaced execution (PR lane only). */ reused?: { inputKey: string; runId: string; revision: string; completedAt: number }; + /** The parent could not run the shard at all (a runner error, never a trial verdict). */ + runnerError?: string; + /** Isolated trial shards only: the trial record this shard produced. */ + trial?: ShardTrialRecord; +} + +/** + * One isolated trial's record, derived from its shard status and the records + * in its own eval dir. `outcome` null means the harness produced no trial + * (never started, hollow, isolation broken, runner error): the panel is then + * INCOMPLETE and the slice exits non-zero. A failed, timed-out or crashed + * trial is a trial verdict; the slice still exits zero and the report decides. + */ +export interface ShardTrialRecord { + case: string; + trial: number; + kind: EvalCaseKind; + panel: PanelShape; + quarantined: boolean; + outcome: TrialOutcome | null; + harness?: string; + failure_class?: TrialFailureClass; + exit_reason?: string; + error?: string; + timeout_at_turn?: number; + cost_usd: number; + duration_ms: number; + model?: string; +} + +/** Records and contract evidence an isolated shard left in its eval dir. */ +export function readTrialEvidence(evalDir: string | undefined): { records: any[]; contract: string | null } { + if (!evalDir || !fs.existsSync(evalDir)) return { records: [], contract: null }; + const names = fs.readdirSync(evalDir); + const parse = (name: string) => { try { return JSON.parse(fs.readFileSync(path.join(evalDir, name), 'utf8')); } catch { return null; } }; + const finalized = names.filter(name => isFinalizedEvalResultFile(name) && !name.startsWith('e2e-reused-')).map(parse).filter(Boolean); + const source = finalized.length ? finalized : names.filter(name => name.startsWith('_partial') && name.endsWith('.json')).map(parse).filter(Boolean); + const records = source.flatMap((result: any) => Array.isArray(result?.tests) ? result.tests.filter((t: any) => t && typeof t === 'object') : []); + let contract: string | null = records.find((t: any) => t.failure_class === 'contract')?.error ?? null; + if (records.some((t: any) => t.failure_class === 'contract') && contract === null) contract = 'contract violation'; + try { + const line = fs.readFileSync(path.join(evalDir, CONTRACT_VIOLATIONS_FILE), 'utf8').split('\n').find(l => l.trim()); + if (line) contract = String(JSON.parse(line).message ?? 'contract violation'); + } catch { /* no sidecar */ } + return { records, contract }; +} + +/** Classify one isolated trial shard. Contract evidence always fails the trial. */ +export function classifyTrialShard( + outcome: Pick, + caseId: string, trial: number, plan: CaseTrialPlan, + evidence: { records: any[]; contract: string | null }, +): ShardTrialRecord { + const failedRecord = evidence.records.find(record => record.passed === false) ?? evidence.records[0]; + const base: ShardTrialRecord = { + case: caseId, trial, kind: plan.kind, panel: plan.panel, quarantined: plan.quarantined, outcome: null, + cost_usd: Math.round(evidence.records.reduce((sum, record) => sum + (Number(record.cost_usd) || 0), 0) * 100) / 100, + duration_ms: outcome.elapsedMs, + ...(typeof failedRecord?.model === 'string' ? { model: failedRecord.model } : {}), + }; + const failed = (failureClass: TrialFailureClass, error?: string): ShardTrialRecord => ({ + ...base, outcome: 'failed', failure_class: evidence.contract !== null ? 'contract' : failureClass, + ...(failedRecord?.exit_reason ? { exit_reason: String(failedRecord.exit_reason) } : {}), + ...(Number.isInteger(failedRecord?.timeout_at_turn) ? { timeout_at_turn: failedRecord.timeout_at_turn } : {}), + ...(sanitizeTrialError(evidence.contract ?? failedRecord?.error ?? error) ? { error: sanitizeTrialError(evidence.contract ?? failedRecord?.error ?? error) } : {}), + }); + if (outcome.runnerError !== undefined) return { ...base, harness: `runner error: ${sanitizeTrialError(outcome.runnerError) ?? 'unknown'}` }; + if (outcome.status === 'never-started') return { ...base, harness: 'never started' }; + if (outcome.status === 'passed-empty') return { ...base, harness: 'hollow: executed no case' }; + if (outcome.status === 'skipped-by-diff') return { ...base, harness: 'skipped by diff' }; + const known = outcome.executedTests !== null && outcome.skippedTests !== null; + const ran = known ? outcome.executedTests! - outcome.skippedTests! : null; + if (outcome.status === 'timed-out') { + if (ran !== null && ran > 1) return { ...base, harness: `isolation broken: ${ran} cases ran` }; + return failed('timeout', 'shard wall reached'); + } + if (ran === null) return outcome.status === 'failed' ? failed('infra', 'crashed without a test summary') : { ...base, harness: 'no test summary' }; + if (ran > 1) return { ...base, harness: `isolation broken: ${ran} cases ran` }; + if (ran === 0) { + if (outcome.status === 'passed' && outcome.skippedTests! > 0 && evidence.contract === null) return { ...base, outcome: 'skipped' }; + if (outcome.status === 'failed') return failed('infra', 'the case never ran (load or setup failure)'); + return { ...base, harness: 'hollow: executed no case' }; + } + if (outcome.status === 'passed') return evidence.contract !== null ? failed('contract') : { ...base, outcome: 'passed' }; + return failed(failedRecord ? failureClassOf(failedRecord) : 'assertion'); } /** @@ -775,6 +969,8 @@ export interface RunShardsOptions { expectedCaseIds?: Record; /** PR lane only: verified reuse for one shard's exact child environment and wall. */ reuseFor?: (files: string[], env: NodeJS.ProcessEnv, budget: PaidShardBudget) => E2EShardReuse | null; + /** Isolated trial shards: key -> the case's fixed trial plan. */ + trials?: Record; } let shardLogSequence = 0; @@ -832,6 +1028,23 @@ export async function runPaidShard( if (options.evalDirBase) { env.GSTACK_EVAL_DIR = path.join(options.evalDirBase, 'shards', shardSlug(files)); } + const trialPlan = files.length === 1 ? options.trials?.[normalizeRelativePath(files[0]!)] : undefined; + const trialIndex = files.length === 1 ? shardTrial(files[0]!) : null; + if (trialPlan && trialIndex !== null && caseId !== null) { + // One case, one trial: the selection binds the child to exactly this id, + // and eval-store stamps every record with the trial identity. + Object.assign(env, { + [TRIAL_ENV.caseId]: caseId, [TRIAL_ENV.kind]: trialPlan.kind, [TRIAL_ENV.trial]: String(trialIndex), + [TRIAL_ENV.panelN]: String(trialPlan.panel.n), [TRIAL_ENV.panelK]: String(trialPlan.panel.k), + [TRIAL_ENV.policyVersion]: String(EVAL_POLICY.version), + EVALS_SELECTION_JSON: JSON.stringify({ version: 1, selected: [caseId], reason: `trial ${trialIndex}/${trialPlan.panel.n} of ${caseId}` }), + }); + } else { + for (const name of Object.values(TRIAL_ENV)) delete env[name]; + } + const withTrial = (outcome: ShardOutcome): ShardOutcome => trialPlan && trialIndex !== null && caseId !== null + ? { ...outcome, trial: classifyTrialShard(outcome, caseId, trialIndex, trialPlan, readTrialEvidence(env.GSTACK_EVAL_DIR)) } + : outcome; // Resolve `claude --version` ONCE in the parent (cached across shards) and // hand it to every child: eval-store's fallback is a synchronous spawn on // the same thread that polls PTY sessions, so children must never pay it. @@ -859,9 +1072,9 @@ export async function runPaidShard( }, null, 2)}\n`); } log(`${label} REUSED ${files.join(' ')} — identical inputs passed in run ${reused.source.runId} at ${reusedFrom.completed_at}`); - return { shard: shardNumber, files, status: 'passed', exitCode: 0, elapsedMs: 0, groupPid: null, + return withTrial({ shard: shardNumber, files, status: 'passed', exitCode: 0, elapsedMs: 0, groupPid: null, executedTests: caseIds.length, skippedTests: 0, budget, - reused: { inputKey: reused.key, runId: reused.source.runId, revision: reused.source.revision, completedAt: reused.source.completedAt } }; + reused: { inputKey: reused.key, runId: reused.source.runId, revision: reused.source.revision, completedAt: reused.source.completedAt } }); } const { command, args } = options.commandFor ? options.commandFor(files) @@ -872,8 +1085,11 @@ export async function runPaidShard( timeoutMs, options.withinShardConcurrency ?? DEFAULT_WITHIN_SHARD_CONCURRENCY, retriesForFiles(files), - ), ...(casePattern !== undefined ? ['--test-name-pattern', casePattern] : [])], + ), ...(casePattern !== undefined ? ['--test-name-pattern', casePattern] : []), + // Per-test outcomes for pass-rate history, keyed by Bun test name. + ...(env.GSTACK_EVAL_DIR ? ['--reporter=junit', '--reporter-outfile', path.join(env.GSTACK_EVAL_DIR, 'junit.xml')] : [])], }; + if (env.GSTACK_EVAL_DIR) fs.mkdirSync(env.GSTACK_EVAL_DIR, { recursive: true }); // Per-shard temp + Chromium-profile isolation — the free runner treats // this as mandatory (test-free-shards.ts: two concurrent shards on one // profile dir kill each other's browser; shared tmp cross-contaminates), @@ -1057,7 +1273,7 @@ export async function runPaidShard( ? summary.terminalTestCounts.reduce((a, b) => a + b, 0) : null; const skippedTests = summary.terminalTestCounts.length > 0 ? summary.skippedTests : null; - return { shard: shardNumber, files, status, exitCode, elapsedMs, groupPid, executedTests, skippedTests, budget }; + return withTrial({ shard: shardNumber, files, status, exitCode, elapsedMs, groupPid, executedTests, skippedTests, budget }); } export interface RunSummary { @@ -1166,7 +1382,8 @@ export async function runPaidShards( try { outcomes[index] = await runPaidShard(shards[index], index + 1, shards.length, { ...options, jobs }); } catch (error) { - outcomes[index] = { + const runnerError = error instanceof Error ? error.message : String(error); + const failed: ShardOutcome = { shard: index + 1, files: shards[index], status: 'failed', @@ -1175,7 +1392,13 @@ export async function runPaidShards( groupPid: null, executedTests: null, skippedTests: null, + runnerError, }; + const key = shards[index].length === 1 ? normalizeRelativePath(shards[index][0]!) : ''; + const plan = options.trials?.[key]; + outcomes[index] = plan && shardCaseId(key) !== null && shardTrial(key) !== null + ? { ...failed, trial: classifyTrialShard(failed, shardCaseId(key)!, shardTrial(key)!, plan, { records: [], contract: null }) } + : failed; console.error(`[test:paid] shard ${index + 1} could not run: ${error instanceof Error ? error.message : String(error)}`); } finally { if (overlay) activeOverlayShards--; @@ -1229,6 +1452,10 @@ export interface ManifestEntry { budget?: PaidShardBudget; /** Budget-mode packing weight (recorded wall, or the whole budget when unknown). */ estimatedMs?: number; + /** Isolated trial shard (`#~t`): the case's kind, fixed panel and quarantine at plan time. */ + trial?: CaseTrialPlan; + /** File shard whose isolated cases run as trial shards: their ids, excluded by name here. */ + excludeCases?: string[]; } /** Budget-mode plan: per-executor estimate and the CI job timeout it needs. */ @@ -1293,18 +1520,48 @@ export function writePaidTestDurations(tier: PaidTier, durations: Record#`). */ +function durationKey(key: string): string { + const rel = normalizeRelativePath(key); + return shardTrial(rel) === null ? rel : `${shardFile(rel)}${CASE_KEY_SEPARATOR}${shardCaseId(rel)}`; +} + +/** Recorded wall of a shard key; an unrecorded trial falls back to its whole file's wall. */ +export function recordedShardMs(recorded: Record, key: string): number | undefined { + const rel = normalizeRelativePath(key); + return recorded[durationKey(rel)] ?? (shardTrial(rel) === null ? undefined : recorded[shardFile(rel)]); +} + +/** + * Merge a report's executed single-file outcomes into the seed; all-skipped + * shards carry no cost signal. Trials of one case record their longest wall + * under the case key. + */ export function mergePaidTestDurations(seed: Record, results: SliceResult[]): Record { const merged = { ...seed }; + const fresh = new Map(); for (const result of results) { for (const outcome of result.outcomes) { - if (outcome.files.length !== 1 || outcome.elapsedMs < 1_000 || isAllSkippedPass(outcome)) continue; - merged[normalizeRelativePath(outcome.files[0])] = outcome.elapsedMs; + if (outcome.files.length !== 1 || outcome.elapsedMs < 1_000 || isAllSkippedPass(outcome) || outcome.reused) continue; + const key = durationKey(outcome.files[0]!); + fresh.set(key, Math.max(fresh.get(key) ?? 0, outcome.elapsedMs)); } } + for (const [key, ms] of fresh) merged[key] = ms; return Object.fromEntries(Object.entries(merged).sort(([a], [b]) => (a < b ? -1 : 1))); } +/** Panel identity of a trial shard key (`#`), else null. */ +export function trialPanelKey(key: string): string | null { + return shardTrial(key) === null ? null : durationKey(key); +} + +/** True when `key` is a trial whose panel already has a trial in `planned`. */ +function sharesPanel(planned: readonly string[], key: string): boolean { + const panel = trialPanelKey(key); + return panel !== null && planned.some(other => other !== key && trialPanelKey(other) === panel); +} + /** Setup, image pull and artifact upload allowance on top of a slice's supervised wall. */ export const CI_SETUP_ALLOWANCE_MINUTES = 20; @@ -1342,12 +1599,15 @@ export function packBySliceBudget(files: string[], budgetMs: number, jobs: numbe recorded: Record, timeoutMs?: number): { slices: string[][]; estimates: Record; estimatedSliceMs: number[]; ciTimeoutMinutes: number; } { - const estimates = Object.fromEntries(files.map(file => [file, recorded[normalizeRelativePath(file)] ?? budgetMs])); + const estimates = Object.fromEntries(files.map(file => [file, recordedShardMs(recorded, file) ?? budgetMs])); const weight = (file: string) => estimates[file]!; const slices: string[][] = []; for (const file of sliceExecutionOrder(files.filter(file => !isOverlayTestFile(file)).map(file => ({ file, estimatedMs: weight(file) }))).map(entry => entry.file)) { let best = -1, bestMs = -1; slices.forEach((planned, index) => { + // Trials of one case never share a runner: independent machines, and + // the panel's wall stays one trial long. + if (sharesPanel(planned, file)) return; const ms = estimatedSliceMs([...planned, file], weight, jobs); if (ms <= budgetMs && ms > bestMs) { best = index; bestMs = ms; } }); @@ -1396,6 +1656,9 @@ export function buildRunManifest(opts: { durations?: Record; /** Weekly gate census only: LLM judges already run in the periodic census and PR gate lanes. */ skipJudges?: boolean; + /** Injectable registries (default: E2E_KINDS and CASE_QUARANTINE). */ + kinds?: Record; + quarantine?: Record; }): PaidRunManifest { const budgetMode = opts.sliceBudgetMs !== undefined; if (budgetMode === (opts.sliceCount !== undefined)) throw new Error('Plan with exactly one of --slices or --slice-budget'); @@ -1413,27 +1676,38 @@ export function buildRunManifest(opts: { const tierSelection = selectPaidTestFiles(discovered, opts.tier, rootDir, env); const judge = (file: string) => /^test\/skill-llm-eval[^/]*\.test\.ts$/.test(normalizeRelativePath(file)); const selected = opts.skipJudges ? tierSelection.selected.filter(file => !judge(file)) : tierSelection.selected; + const kinds = opts.kinds ?? E2E_KINDS; + const quarantine = opts.quarantine ?? CASE_QUARANTINE; + const notLive = [...Object.keys(kinds).filter(id => kinds[id] === 'behavior'), ...Object.keys(quarantine)].filter(id => !Object.hasOwn(E2E_TIERS, id)); + if (notLive.length) throw new Error(`Only live E2E cases can be behavior or quarantined (judges sample their panel inside the case): ${notLive.join(', ')}`); const caseKeys = partitionCaseExclusions(expandCaseShards(selected, opts.tier, rootDir)); const excluded = [...tierSelection.excluded, ...(opts.skipJudges ? tierSelection.selected.filter(judge) .map(file => ({ file, reason: 'skipped: LLM judges run in the periodic census and PR gate lanes' })) : []), ...caseKeys.excluded]; - const shards = planPaidShards(caseKeys.runnable, { maxFilesPerShard: 1 }); + const expansion = expandTrialShards(caseKeys.runnable, opts.tier, rootDir, { kinds, quarantine }); + const excludeOf = (key: string) => expansion.excludeCases[normalizeRelativePath(key)] ?? []; + const shards = planPaidShards(expansion.keys, { maxFilesPerShard: 1 }); const cases = computePaidCaseSelection({ profile, env, rootDir, changedFiles: opts.changedFiles }); const fast = cases.coverage?.mode === 'pr'; - const profileShards = fast ? shards.filter(files => prProfileFileSelected(files[0], cases.selection)) : shards; + const profileShards = fast ? shards.filter(files => prProfileFileSelected(files[0], cases.selection, excludeOf(files[0]!))) : shards; const { runnable, skipped } = partitionShardsByDiffSelection(profileShards, - cases.selection.e2e === null ? null : new Set(cases.selection.e2e), { rootDir }); + cases.selection.e2e === null ? null : new Set(cases.selection.e2e), { rootDir, excludeCases: expansion.excludeCases }); if (fast) for (const files of shards) { - if (!prProfileFileSelected(files[0], cases.selection)) skipped.push({ files, reason: 'Outside the fast PR profile; retained in broad gate/periodic coverage' }); + if (!prProfileFileSelected(files[0], cases.selection, excludeOf(files[0]!))) skipped.push({ files, reason: 'Outside the fast PR profile; retained in broad gate/periodic coverage' }); } + const extras = (key: string): Pick => { + const trial = expansion.trials[normalizeRelativePath(key)]; + const exclude = expansion.excludeCases[normalizeRelativePath(key)]; + return { ...(trial ? { trial } : {}), ...(exclude ? { excludeCases: exclude } : {}) }; + }; const entries: ManifestEntry[] = []; if (budgetMode) { const plan = packBySliceBudget(runnable.map(files => files[0]!), opts.sliceBudgetMs!, opts.jobs!, opts.durations ?? loadPaidTestDurations(rootDir, opts.tier), opts.timeoutMs); plan.slices.forEach((files, index) => files.forEach(file => entries.push({ file, slice: index + 1, status: 'planned', - estimatedMs: plan.estimates[file]!, + estimatedMs: plan.estimates[file]!, ...extras(file), ...(FILE_RETRY_BUDGETS.some(budget => budget.file === shardFile(file)) ? { budget: resolvePaidShardBudget([file], opts.timeoutMs) } : {}) }))); - for (const s of skipped) entries.push({ file: s.files[0], slice: 0, status: 'skipped-by-diff', reason: s.reason }); + for (const s of skipped) entries.push({ file: s.files[0], slice: 0, status: 'skipped-by-diff', reason: s.reason, ...extras(s.files[0]!) }); for (const e of excluded) entries.push({ file: e.file, slice: 0, status: 'excluded', reason: e.reason }); entries.sort((a, b) => (a.file < b.file ? -1 : 1)); return parseRunManifest(JSON.stringify({ @@ -1464,15 +1738,28 @@ export function buildRunManifest(opts: { for (const files of [...registered].sort(byWall).concat( ordinary.filter(files => !registeredFiles.has(files[0])))) { const lanes = registeredFiles.has(files[0]) ? longLanes : ordinarySlices; - let lane = 0; - for (let index = 1; index < lanes; index++) if (loads[index] < loads[lane]) lane = index; + const laneKeys = (index: number) => [...allocations].filter(([, lane]) => lane === index + 1).map(([key]) => key); + let lane = -1; + for (let index = 0; index < lanes; index++) { + if (sharesPanel(laneKeys(index), files[0]!)) continue; + if (lane < 0 || loads[index] < loads[lane]) lane = index; + } + if (lane < 0) lane = loads.slice(0, lanes).indexOf(Math.min(...loads.slice(0, lanes))); allocations.set(files[0], lane + 1); loads[lane] += resolvePaidShardTimeoutMs(files, opts.timeoutMs); } } let ordinaryIndex = 0; for (const files of ordinary) { - if (!allocations.has(files[0])) allocations.set(files[0], (ordinaryIndex++ % ordinarySlices) + 1); + if (allocations.has(files[0])) continue; + // Round-robin, skipping a lane that already holds a trial of the same panel. + let lane = ordinaryIndex % ordinarySlices; + for (let step = 0; step < ordinarySlices; step++) { + const candidate = (ordinaryIndex + step) % ordinarySlices; + if (!sharesPanel([...allocations].filter(([, l]) => l === candidate + 1).map(([key]) => key), files[0]!)) { lane = candidate; break; } + } + ordinaryIndex++; + allocations.set(files[0], lane + 1); } const packed = packByRecordedDuration(); function packByRecordedDuration(): Map | null { @@ -1483,10 +1770,10 @@ export function buildRunManifest(opts: { ordinary.filter(files => allocations.get(files[0]) === lane + 1).map(files => files[0])); const caps = SUPERVISED_WORKER_COUNTS.map(jobs => Math.max(...lanes.map(files => bound(files, jobs)))); const fits = (files: string[]) => SUPERVISED_WORKER_COUNTS.every((jobs, k) => bound(files, jobs) <= caps[k]); - const known = ordinary.map(files => recorded[normalizeRelativePath(files[0])]) + const known = ordinary.map(files => recordedShardMs(recorded, files[0]!)) .filter((ms): ms is number => ms !== undefined).sort((a, b) => a - b); const fallback = known.length ? known[Math.min(known.length - 1, Math.floor(known.length * 0.75))] : 1; - const weight = (file: string) => recorded[normalizeRelativePath(file)] ?? fallback; + const weight = (file: string) => recordedShardMs(recorded, file) ?? fallback; const load = (files: string[]) => files.reduce((sum, file) => sum + weight(file), 0); const registeredFiles = new Set(registered.map(files => files[0])); // Local search from the supervised baseline: move or swap a file out of @@ -1510,6 +1797,7 @@ export function buildRunManifest(opts: { const heavyAfter = lanes[heavy].filter(file => file !== a).concat(b === null ? [] : [b]); const otherAfter = lanes[other].filter(file => file !== b).concat([a]); if (!fits(heavyAfter) || !fits(otherAfter)) continue; + if (sharesPanel(otherAfter, a) || (b !== null && sharesPanel(heavyAfter, b))) continue; const [h, o] = [heavy, other]; best = { gain, apply: () => { lanes[h] = heavyAfter; lanes[o] = otherAfter; } }; } @@ -1522,11 +1810,11 @@ export function buildRunManifest(opts: { } runnable.forEach((files) => { const slice = files.some(isOverlayTestFile) ? overlaySlice : (packed ?? allocations).get(files[0])!; - entries.push({ file: files[0], slice, status: 'planned', + entries.push({ file: files[0], slice, status: 'planned', ...extras(files[0]!), ...(FILE_RETRY_BUDGETS.some(budget => budget.file === shardFile(files[0]!)) ? { budget: resolvePaidShardBudget(files, opts.timeoutMs) } : {}) }); }); - for (const s of skipped) entries.push({ file: s.files[0], slice: 0, status: 'skipped-by-diff', reason: s.reason }); + for (const s of skipped) entries.push({ file: s.files[0], slice: 0, status: 'skipped-by-diff', reason: s.reason, ...extras(s.files[0]!) }); for (const e of excluded) entries.push({ file: e.file, slice: 0, status: 'excluded', reason: e.reason }); entries.sort((a, b) => (a.file < b.file ? -1 : 1)); @@ -1603,26 +1891,70 @@ export function parseRunManifest(raw: string): PaidRunManifest { if (entry.estimatedMs !== undefined && (entry.status !== 'planned' || !Number.isSafeInteger(entry.estimatedMs) || entry.estimatedMs < 0)) { throw new Error(`manifest entry ${entry.file} has an invalid estimate`); } - if (entry.status === 'planned' && parsed.prCoverage?.mode === 'pr' && !prProfileFileSelected(entry.file, parsed.selection!)) { + if (entry.status === 'planned' && parsed.prCoverage?.mode === 'pr' && !prProfileFileSelected(entry.file, parsed.selection!, entry.excludeCases ?? [])) { throw new Error(`manifest file is outside its PR case selection: ${entry.file}`); } } for (const entry of parsed.entries) { const caseId = shardCaseId(entry.file); - if (caseId === null ? entry.status === 'planned' && CASE_SHARDED_FILES.includes(shardFile(entry.file)) + const trial = shardTrial(entry.file); + if (trial !== null) { + const plan = entry.trial; + const expected = plan && ['rule', 'behavior', 'judge'].includes(plan.kind) && typeof plan.quarantined === 'boolean' + ? caseTrialPlan(caseId!, { [caseId!]: plan.kind }, plan.quarantined ? { [caseId!]: true } : {}) : undefined; + if (!Object.hasOwn(E2E_TOUCHFILES, caseId!) || !E2E_TOUCHFILES[caseId!]!.includes(shardFile(entry.file)) + || !plan || !expected || !isIsolatedCase(expected) || !sameTrialPlan(plan, expected) || trial > plan.panel.n) { + throw new Error(`Trial shard must name a registered isolated case of its file with its fixed policy panel: ${entry.file}`); + } + } else if (entry.trial !== undefined) { + throw new Error(`Only trial shards carry a trial plan: ${entry.file}`); + } else if (caseId === null ? entry.status === 'planned' && CASE_SHARDED_FILES.includes(shardFile(entry.file)) : !CASE_SHARDED_FILES.includes(shardFile(entry.file)) || !(caseId in E2E_TOUCHFILES)) { throw new Error(`Case-sharded files plan one registered case per shard: ${entry.file}`); } + if (entry.excludeCases !== undefined && (caseId !== null || !Array.isArray(entry.excludeCases) || entry.excludeCases.length === 0 + || entry.excludeCases.some(id => !parsed.entries.some(other => shardTrial(other.file) !== null + && shardFile(other.file) === shardFile(entry.file) && shardCaseId(other.file) === id)))) { + throw new Error(`A file shard may exclude only cases that run as its trial shards: ${entry.file}`); + } if (caseId !== null && entry.status === 'planned' && parsed.selection?.e2e && !parsed.selection.e2e.includes(caseId)) { throw new Error(`Planned case shard is outside the manifest selection: ${entry.file}`); } } + // Panels are whole: exactly n trial entries per isolated case with one plan + // and one status, and planned trials on distinct slices when the ordinary + // slices allow it. + const panels = new Map(); + for (const entry of parsed.entries) { + const panel = trialPanelKey(entry.file); + if (panel !== null) panels.set(panel, [...(panels.get(panel) ?? []), entry]); + } + const reservedOverlay = parsed.sliceCount > 1 && parsed.entries.some(entry => entry.status === 'planned' && isOverlayTestFile(entry.file)); + const ordinarySliceCount = parsed.sliceCount - Number(reservedOverlay); + for (const [panel, trials] of panels) { + const n = trials[0]!.trial!.panel.n; + const indices = trials.map(entry => shardTrial(entry.file)!).sort((a, b) => a - b); + if (indices.length !== n || indices.some((index, i) => index !== i + 1) + || trials.some(entry => !sameTrialPlan(entry.trial, trials[0]!.trial) || entry.status !== trials[0]!.status) + || parsed.entries.some(entry => normalizeRelativePath(entry.file) === panel)) { + throw new Error(`Isolated case ${panel} must plan exactly its ${n} trials together`); + } + const slices = trials.filter(entry => entry.status === 'planned').map(entry => entry.slice); + if (ordinarySliceCount >= n && new Set(slices).size !== slices.length) { + throw new Error(`Trials of ${panel} share a slice; each trial needs its own runner`); + } + } if (parsed.prCoverage?.mode === 'pr') { const planned = parsed.entries.filter(entry => entry.status === 'planned').map(entry => normalizeRelativePath(entry.file)); const required: string[][] = Object.entries(PR_PROFILE_FILES).flatMap(([file, ids]) => { const selected = ids.filter(id => parsed.selection!.e2e!.includes(id)); if (!selected.length) return []; - return [CASE_SHARDED_FILES.includes(file) ? selected.map(id => `${file}#${id}`) : [file]]; + const owners = new Set(selected.flatMap(id => { + const trials = parsed.entries.filter(entry => shardTrial(entry.file) !== null && shardFile(entry.file) === file && shardCaseId(entry.file) === id); + if (trials.length) return trials.map(entry => normalizeRelativePath(entry.file)); + return [CASE_SHARDED_FILES.includes(file) ? `${file}#${id}` : file]; + })); + return [[...owners]]; }); if (parsed.selection!.judges!.length) required.push(['test/skill-llm-eval.test.ts']); for (const owners of required) { @@ -1644,6 +1976,14 @@ export function parseRunManifest(raw: string): PaidRunManifest { } const keys = parsed.entries.map(entry => normalizeRelativePath(entry.file)); if (new Set(keys).size !== keys.length) throw new Error('Duplicate manifest entry'); + // Unique result slugs: shard artifacts merge by path, so a shared slug would + // let one trial's records overwrite another's. + const slugs = new Map(); + for (const entry of parsed.entries) { + const slug = shardSlug([entry.file]); + if (slugs.has(slug)) throw new Error(`Shards ${slugs.get(slug)} and ${entry.file} share the result slug ${slug}`); + slugs.set(slug, entry.file); + } const overlaySlice = parsed.sliceCount; const plannedOverlays = parsed.entries.filter(entry => entry.status === 'planned' && isOverlayTestFile(entry.file)); if (plannedOverlays.some(entry => entry.slice !== overlaySlice)) { @@ -1655,7 +1995,9 @@ export function parseRunManifest(raw: string): PaidRunManifest { } for (const budget of FILE_RETRY_BUDGETS) { const entries = parsed.entries.filter(entry => shardFile(entry.file) === budget.file); - if (entries.length > 1 && entries.some(entry => shardCaseId(entry.file) === null)) throw new Error(`Duplicate registered manifest entry: ${budget.file}`); + const fileKeys = entries.filter(entry => shardCaseId(entry.file) === null).length; + const caseKeys = entries.filter(entry => shardCaseId(entry.file) !== null && shardTrial(entry.file) === null).length; + if (fileKeys > 1 || (fileKeys === 1 && caseKeys > 0)) throw new Error(`Duplicate registered manifest entry: ${budget.file}`); for (const entry of entries.filter(entry => entry.status === 'planned')) { if (!entry.budget) throw new Error(`Registered manifest needs an explicit budget record: ${budget.file}`); const expected = resolvePaidShardBudget([entry.file], entry.budget.source === 'explicit' ? entry.budget.timeoutMs : undefined); @@ -1673,7 +2015,31 @@ export interface SliceResult { sliceIndex: number; sliceCount: number; timeoutOverrideMs?: number; - outcomes: Array>; + /** CI run attempt (github.run_attempt) that produced this slice; absent means 1. */ + attempt?: number; + /** Epoch ms bounds of the slice's shard execution (lane wall time). */ + startedAt?: number; + finishedAt?: number; + outcomes: Array>; +} + +/** + * Slice exit = execution completeness, never the semantic verdict. A rule + * shard that did not pass fails the slice (unchanged fail-closed rule); an + * isolated trial shard fails it only when the harness produced no trial + * record. Failed, timed-out or crashed trials are verdict input for the + * report's panelVerdict(), so a 2/3 PASS panel never reds its runner. + */ +export function sliceExitCode(outcomes: ReadonlyArray>): number { + return outcomes.every(outcome => outcome.trial !== undefined + ? outcome.trial.outcome !== null + : outcome.status === 'passed' || outcome.status === 'skipped-by-diff') ? 0 : 1; +} + +/** A hollow-guarded trial shard has no trial record: the guard's verdict is a harness problem. */ +export function guardTrialRecords>(outcomes: T[]): T[] { + return outcomes.map(outcome => outcome.trial && outcome.status === 'passed-empty' && outcome.trial.outcome !== null + ? { ...outcome, trial: { ...outcome.trial, outcome: null, harness: 'hollow: executed no case' } } : outcome); } /** @@ -1704,21 +2070,35 @@ export function verifySliceResults( if (!byIndex.has(index)) problems.push(`slice ${index}/${manifest.sliceCount} reported NO result — cancelled/crashed executor, not a pass`); } - const reported = new Map(); + const reported = new Map(); + const planned = new Map(manifest.entries.map(entry => [normalizeRelativePath(entry.file), entry])); for (const result of results) { for (const outcome of result.outcomes) { if (outcome.files.some(file => FILE_RETRY_BUDGETS.some(budget => budget.file === shardFile(file)) || shardCaseId(file) !== null) && outcome.files.length !== 1) { problems.push('Registered result must report its own shard'); } const file = normalizeRelativePath(outcome.files[0] ?? ''); + const trialPlan = planned.get(file)?.trial; if (manifest.prCoverage?.mode === 'pr') { - const expected = expectedPrCaseCount(file, manifest.selection!); + const expected = expectedPrCaseCount(file, manifest.selection!, planned.get(file)?.excludeCases); const executed = outcome.executedTests === null || outcome.skippedTests === null ? -1 : outcome.executedTests - outcome.skippedTests; - if (outcome.exitCode !== 0 || expected < 1 || executed !== expected) { + // A trial's exit status is its verdict (panelVerdict decides); its + // completeness is the trial record checked below. + if ((!trialPlan && outcome.exitCode !== 0) || expected < 1 || (!trialPlan && executed !== expected)) { problems.push(`PR profile expected ${expected} executed cases in ${file}, received ${executed}`); } } + if (trialPlan) { + const t = outcome.trial; + if (!t || t.case !== shardCaseId(file) || t.trial !== shardTrial(file) || !sameTrialPlan(t, trialPlan) + || !(t.outcome === null || ['passed', 'failed', 'skipped'].includes(t.outcome)) + || (t.outcome === 'failed') !== (t.failure_class !== undefined)) { + problems.push(`${file}: trial record missing or does not match its planned trial`); + } + } else if (outcome.trial !== undefined) { + problems.push(`${file}: an unplanned trial record`); + } if (shardCaseId(file) !== null && outcome.status === 'passed' && (outcome.executedTests === null || outcome.skippedTests === null || outcome.executedTests - outcome.skippedTests !== 1)) { problems.push(`Case shard must execute exactly its one case: ${file}`); @@ -1732,7 +2112,7 @@ export function verifySliceResults( } } if (reported.has(file)) problems.push(`${file} reported by two slices`); - reported.set(file, { slice: result.sliceIndex, status: outcome.status }); + reported.set(file, { slice: result.sliceIndex, status: outcome.status, ...(outcome.trial ? { trial: outcome.trial } : {}) }); const registered = FILE_RETRY_BUDGETS.find(budget => budget.file === shardFile(file)); const finding = STRICT_RETRY_CASE_BUDGETS.find(budget => budget.file === file); if (finding) { @@ -1762,7 +2142,10 @@ export function verifySliceResults( continue; // the missing-slice problem above already covers it } if (got.slice !== entry.slice) problems.push(`${entry.file} planned for slice ${entry.slice} but reported by slice ${got.slice}`); - if (got.status !== 'passed') problems.push(`${entry.file}: ${got.status}`); + // Isolated trial shards: harness health only; the panel verdict gates. + if (entry.trial) { + if (got.trial?.outcome === null) problems.push(`${entry.file}: no trial record (${got.trial.harness ?? 'unknown'})`); + } else if (got.status !== 'passed') problems.push(`${entry.file}: ${got.status}`); } return { ok: problems.length === 0, problems }; } @@ -1783,6 +2166,27 @@ export function formatSlicePlan(manifest: PaidRunManifest): string[] { return lines; } +/** + * Capacity preflight (E-A10): what the plan asks of the runner pool and the + * API. Sessions are planned shard processes; at most jobs of them run per + * slice at once. Waves > 1 mean slices queue behind the matrix cap and the + * lane wall grows by whole slices. + */ +export function formatCapacityPreflight(manifest: PaidRunManifest, maxParallel?: number): string[] { + const planned = manifest.entries.filter(entry => entry.status === 'planned'); + const trials = planned.filter(entry => entry.trial); + const jobs = manifest.plan?.jobs ?? DEFAULT_JOBS; + const longest = [...trials].sort((a, b) => (b.estimatedMs ?? 0) - (a.estimatedMs ?? 0))[0]; + const waves = maxParallel ? Math.ceil(manifest.sliceCount / maxParallel) : null; + return [ + `[test:paid] capacity: ${manifest.sliceCount} slice(s), ${planned.length} planned shard(s) (${planned.length - trials.length} rule/judge, ${trials.length} trial shard(s) in ${new Set(trials.map(entry => trialPanelKey(entry.file))).size} panel(s)); ` + + `peak ${Math.min(manifest.sliceCount, maxParallel ?? manifest.sliceCount) * jobs} concurrent shard process(es)` + + (waves !== null ? `; wave(s) at max-parallel ${maxParallel}: ${waves}` : ''), + ...(longest ? [`[test:paid] capacity: longest indivisible trial ~${((longest.estimatedMs ?? 0) / 60_000).toFixed(1)}m (${longest.file})`] : []), + ...(waves !== null && waves > 1 ? [`[test:paid] capacity: ⚠ ${manifest.sliceCount} slices exceed max-parallel ${maxParallel}; later slices queue for a second wave`] : []), + ]; +} + export function formatProfileCoverage(manifest: PaidRunManifest): string[] { const coverage = manifest.prCoverage; return [ @@ -1791,19 +2195,15 @@ export function formatProfileCoverage(manifest: PaidRunManifest): string[] { ]; } -/** Final outcomes use each case's last attempt; the attempt total stays visible. */ +/** Every collector record counts: paid evals never retry, so a later record never replaces an earlier one. */ export function collectorOutcomeCounts(results: Array<{ tests?: Array<{ name: string; suite?: string; passed: boolean; execution?: string; manual_review?: unknown; }> }>): { executed: number; reused: number; passed: number; failed: number; manual_accepted: number; attempts: number } { const counts = { executed: 0, reused: 0, passed: 0, failed: 0, manual_accepted: 0, attempts: 0 }; for (const result of results) { - const cases = new Map[number]>(); for (const entry of result.tests ?? []) { if (!entry || typeof entry !== 'object' || typeof entry.name !== 'string' || typeof entry.passed !== 'boolean') continue; counts.attempts++; - cases.set(`${entry.suite ?? ''}\0${entry.name}`, entry); - } - for (const entry of cases.values()) { counts[entry.execution === 'reused' ? 'reused' : 'executed']++; const outcome = evalEntryOutcome(entry); counts[outcome === 'manual-review' ? 'manual_accepted' : outcome]++; @@ -1812,6 +2212,438 @@ export function collectorOutcomeCounts(results: Array<{ tests?: Array<{ return counts; } + +// ─── Report: verdicts, history records and the human readout ─────────────── + +/** One Bun JUnit testcase (`--reporter=junit`). */ +export interface JUnitCase { name: string; classname: string; outcome: TrialOutcome; timeMs: number; failureType?: string; message?: string } + +const xmlUnescape = (text: string) => text.replace(/&(lt|gt|quot|apos|amp|#(\d+)|#x([0-9a-f]+));/gi, (_, name: string, dec?: string, hex?: string) => + dec ? String.fromCodePoint(Number(dec)) : hex ? String.fromCodePoint(parseInt(hex, 16)) + : ({ lt: '<', gt: '>', quot: '"', apos: "'", amp: '&' } as Record)[name.toLowerCase()]!); + +function xmlAttributes(tag: string): Record { + return Object.fromEntries([...tag.matchAll(/([\w:-]+)="([^"]*)"/g)].map(match => [match[1]!, xmlUnescape(match[2]!)])); +} + +/** Per-test outcomes from a Bun JUnit report; unparseable input yields []. */ +export function parseJUnitCases(xml: string): JUnitCase[] { + const cases: JUnitCase[] = []; + for (const match of xml.matchAll(/]*?)(\/>|>([\s\S]*?)<\/testcase>)/g)) { + const attrs = xmlAttributes(match[1]!); + const body = match[3] ?? ''; + const failure = /<(failure|error)\b([^>]*?)(?:\/>|>)/.exec(body); + const failureAttrs = failure ? xmlAttributes(failure[2]!) : {}; + cases.push({ + name: attrs.name ?? '', classname: attrs.classname ?? '', + outcome: failure ? 'failed' : / CASE_TEST_NAMES[id] === name) ?? null; +} + +interface ReportArtifact { root: string; result: SliceResult } + +/** Every slice result under the report dir: flat (merged) or one directory per attempt-scoped artifact. */ +export function loadSliceArtifacts(reportDir: string): ReportArtifact[] { + const found: ReportArtifact[] = []; + for (const name of fs.readdirSync(reportDir, { recursive: true }) as string[]) { + const rel = normalizeRelativePath(name); + if (!/^slice-\d+\.json$/.test(path.basename(rel)) || rel.split('/').includes('shards') || rel.split('/').includes('receipts')) continue; + found.push({ root: path.join(reportDir, path.dirname(rel)), result: JSON.parse(fs.readFileSync(path.join(reportDir, rel), 'utf8')) as SliceResult }); + } + return found.sort((a, b) => (a.result.attempt ?? 1) - (b.result.attempt ?? 1) || a.result.sliceIndex - b.result.sliceIndex); +} + +export interface PanelReport extends PanelVerdict { + file: string; + /** Slice per trial index (trial n -> slice), for the rerun/artifact pointer. */ + slices: Record; +} + +/** Panel verdicts of one run attempt: exactly the planned trials, each from its reported record. */ +export function panelReports(manifest: PaidRunManifest, results: SliceResult[], attempt: number): PanelReport[] { + const reported = new Map(); + for (const result of results.filter(r => (r.attempt ?? 1) === attempt)) { + for (const outcome of result.outcomes) reported.set(normalizeRelativePath(outcome.files[0] ?? ''), { slice: result.sliceIndex, outcome }); + } + const panels = new Map(); + for (const entry of manifest.entries.filter(e => e.status === 'planned' && e.trial)) { + const key = trialPanelKey(entry.file)!; + panels.set(key, [...(panels.get(key) ?? []), entry]); + } + return [...panels.entries()].sort(([a], [b]) => (a < b ? -1 : 1)).map(([key, entries]) => { + const plan = entries[0]!.trial!; + const slices: Record = {}; + const trials = entries.flatMap(entry => { + const got = reported.get(normalizeRelativePath(entry.file)); + const t = got?.outcome.trial; + if (!got || !t || t.outcome === null) return []; + slices[t.trial] = got.slice; + return [{ trial: t.trial, outcome: t.outcome, attempt, ...(t.failure_class ? { failure_class: t.failure_class } : {}), + ...(t.exit_reason ? { exit_reason: t.exit_reason } : {}), ...(t.error ? { error: t.error } : {}), + ...(got.outcome.reused ? { execution: 'reused' as const } : {}), + ...(t.timeout_at_turn !== undefined ? { timeout_at_turn: t.timeout_at_turn } : {}) }]; + }); + const verdict = panelVerdict({ case: shardCaseId(key)!, kind: plan.kind, panel: plan.panel, trials, quarantined: plan.quarantined }); + return { ...verdict, file: shardFile(key), slices }; + }); +} + +/** Why one failed trial failed, in one line: a case timeout names its turn. */ +function trialCause(trial: PanelVerdict['trials'][number] & { timeout_at_turn?: number }): string { + const cls = trial.failure_class ?? 'assertion'; + const head = cls === 'timeout' || trial.exit_reason === 'timeout' + ? `timeout${trial.timeout_at_turn !== undefined ? ` at turn ${trial.timeout_at_turn}` : ''}` + : cls; + return `t${trial.trial}: ${head}${trial.exit_reason && trial.exit_reason !== 'timeout' ? ` (${trial.exit_reason})` : ''}${trial.error ? ` — ${trial.error}` : ''}`; +} + +export function rerunCommand(tier: PaidTier, id: string, trials: number): string { + return `bun run scripts/test-paid-shards.ts --tier ${tier} --case ${id}${trials > 1 ? ` --trials ${trials}` : ''}`; +} + +/** One line per non-PASS or split panel verdict. */ +export function formatPanelLine(panel: PanelReport, tier: PaidTier): string { + const mark = panel.status === 'PASS' ? '⚠' : panel.failsLane ? '✗' : '◌'; + const label = panel.status === 'PASS' ? `PASS ${panel.passed}/${panel.panel.n}` : `${panel.status} ${panel.passed}/${panel.panel.n}`; + const causes = panel.trials.filter(t => t.outcome !== 'passed').map(t => trialCause(t as any)); + const where = Object.entries(panel.slices).map(([trial, slice]) => `t${trial}@slice ${slice}`).join(', '); + return `${mark} ${panel.case} ${panel.kind}${panel.quarantined ? ' (quarantined)' : ''} ${label} (${panel.marks})` + + `${causes.length ? ` ${causes.join('; ')}` : ''}${panel.status === 'INCOMPLETE' ? ` [${panel.reason}]` : ''}` + + `${where ? ` [${where}, attempt ${panel.attempt}]` : ''} rerun: ${rerunCommand(tier, panel.case, panel.panel.n)}`; +} + +export interface ReportHeadline { + lane: string; + verdict: 'GREEN' | 'RED'; + attempt: number; + counts: { + rule: { passed: number; total: number }; + behavior: { passed: number; total: number; split: number }; + judge: { passed: number; total: number }; + quarantined: { total: number; failingLane: number }; + skipped: number; + infra: number; + incomplete: number; + unattributed: number; + }; + actionRequired: number; + wallMs: number | null; + costUsd: number; + redispatchEligible: boolean; +} + +export function formatHeadline(h: ReportHeadline): string[] { + const c = h.counts; + const minutes = h.wallMs === null ? 'unknown' : `${Math.floor(h.wallMs / 60_000)}m${String(Math.round((h.wallMs % 60_000) / 1000)).padStart(2, '0')}s`; + return [ + `[test:paid] VERDICT ${h.verdict} — lane ${h.lane}, attempt ${h.attempt}`, + ` rule ${c.rule.passed}/${c.rule.total} · behavior ${c.behavior.passed}/${c.behavior.total}${c.behavior.split ? ` (${c.behavior.split} split)` : ''}` + + ` · judge ${c.judge.passed}/${c.judge.total} · quarantined ${c.quarantined.total} (${c.quarantined.failingLane} failing the lane)`, + ` SKIPPED ${c.skipped} · INFRA ${c.infra} · INCOMPLETE ${c.incomplete} · unattributed ${c.unattributed} · ACTION REQUIRED ${h.actionRequired}`, + ` wall ${minutes} · cost $${h.costUsd.toFixed(2)}${h.redispatchEligible ? ' · every red is machine-classified INFRA/INCOMPLETE: eligible for ONE re-dispatch as a new run (EVAL_POLICY.infraRedispatch); report both runs' : ''}`, + ]; +} + +/** Problems a runner loss or an API/CLI failure before grading produces; nothing else qualifies for re-dispatch. */ +const INFRA_PROBLEMS = [ + /^slice \d+\/\d+ reported NO result/, + /^planned .* was never reported$/, + /: never-started$/, + /: no trial record \((?:never started|runner error: .*|no test summary)\)$/, + /^PANEL \S+ INCOMPLETE /, + /^PANEL \S+ FAIL \(INFRA\)/, +]; +export function infraOnly(problems: readonly string[]): boolean { + return problems.length > 0 && problems.every(problem => INFRA_PROBLEMS.some(re => re.test(problem))); +} + +/** + * Report mode: reconcile slice artifacts against the manifest (fail-closed), + * compute every panel verdict with panelVerdict(), write collector-outcomes + * v2, trial-outcomes.jsonl and report-summary.md, and exit non-zero when the + * lane is red. Only the earliest run attempt decides the lane; later attempts + * are reported beside it and never replace it. + */ +export function runPaidReport(reportDir: string, options: { writeDurations?: boolean; env?: NodeJS.ProcessEnv; rootDir?: string } = {}): number { + const env = options.env ?? process.env; + const rootDir = options.rootDir ?? ROOT; + const summaryPath = path.join(reportDir, 'collector-outcomes.json'); + const summaryMdPath = path.join(reportDir, 'report-summary.md'); + const trialOutcomesPath = path.join(reportDir, TRIAL_OUTCOMES_FILE); + for (const file of [summaryPath, summaryMdPath, trialOutcomesPath]) fs.rmSync(file, { force: true }); + const manifest = parseRunManifest(fs.readFileSync(path.join(reportDir, 'manifest.json'), 'utf-8')); + const artifacts = loadSliceArtifacts(reportDir); + const attempts = [...new Set(artifacts.map(a => a.result.attempt ?? 1))].sort((a, b) => a - b); + const primary = attempts[0] ?? 1; + const results = artifacts.filter(a => (a.result.attempt ?? 1) === primary).map(a => a.result); + const verdict = verifySliceResults(manifest, results); + const planned = manifest.entries.filter((e) => e.status === 'planned').length; + const lane = `${manifest.tier}/${manifest.profile ?? 'full'}${manifest.evalsAll ? ' census' : ''}`; + console.log(`[test:paid] report: ${results.length}/${manifest.sliceCount} slices, ${planned} planned shards, tier=${manifest.tier}, attempt ${primary}${attempts.length > 1 ? ` (later attempts ${attempts.slice(1).join(', ')} reported, never replacing it)` : ''}`); + for (const line of formatProfileCoverage(manifest)) console.log(line); + for (const result of [...results].sort((a, b) => a.sliceIndex - b.sliceIndex)) { + for (const outcome of result.outcomes) { + const shown = outcome.reused ? `reused (run ${outcome.reused.runId})` : outcome.trial + ? `trial ${outcome.trial.outcome ?? 'NO RECORD'}` : outcome.status; + console.log(` slice ${result.sliceIndex} ${shown.padEnd(15)} ${String(Math.round(outcome.elapsedMs / 1000)).padStart(5)}s ${outcome.files.join(' ')}`); + } + } + if (options.writeDurations) { + const durations = mergePaidTestDurations(loadPaidTestDurations(rootDir, manifest.tier), results); + writePaidTestDurations(manifest.tier, durations, rootDir); + console.log(`[test:paid] wrote ${Object.keys(durations).length} ${manifest.tier} durations to ${PAID_TEST_DURATIONS_FILE}`); + } + + // Which artifact root (attempt) and which shard (isolated or not) each file belongs to. + const roots = artifacts.map(a => ({ root: path.resolve(a.root), attempt: a.result.attempt ?? 1 })) + .sort((a, b) => b.root.length - a.root.length); + const attemptOf = (abs: string) => roots.find(r => abs === r.root || abs.startsWith(r.root + path.sep))?.attempt ?? primary; + const entryBySlug = new Map(manifest.entries.map(entry => [shardSlug([entry.file]), entry])); + const shardOf = (rel: string) => { + const parts = normalizeRelativePath(rel).split('/'); + const at = parts.lastIndexOf('shards'); + return at >= 0 && parts[at + 1] ? entryBySlug.get(parts[at + 1]!) ?? null : null; + }; + + const flaky: Array<{ name: string; attempts: number; file: string }> = []; + const collectors: Parameters[0] = []; + const files: Array<{ file: string; tier: string; shard: string | number; cost: number; + flaky: number; total: number; executed: number; reused: number; passed: number; + failed: number; manual_accepted: number; attempts: number }> = []; + const manualProblems: string[] = []; + const manualClaims = new Map(); + const recordsByShard = new Map(); + let costUsd = 0; + for (const name of fs.readdirSync(reportDir, { recursive: true }) as string[]) { + const rel = normalizeRelativePath(name); + if (!isFinalizedEvalResultFile(rel) || rel.split('/').includes('receipts') || rel === 'collector-outcomes.json') continue; + if (attemptOf(path.resolve(reportDir, rel)) !== primary) continue; + const shard = shardOf(rel); + try { + const parsed = JSON.parse(fs.readFileSync(path.join(reportDir, rel), 'utf-8')); + if (!Array.isArray(parsed.tests)) { + if (Object.hasOwn(parsed, 'tests') || parsed.total_tests !== undefined || parsed.manual_review !== undefined) { + manualProblems.push(`${rel}: malformed collector tests[]`); + } + continue; + } + costUsd += Number(parsed.total_cost_usd) || 0; + if (shard) recordsByShard.set(shard.file, [...(recordsByShard.get(shard.file) ?? []), ...parsed.tests]); + // Trial records are verdict input for panelVerdict(), never collector gates. + if (shard?.trial) continue; + const seen = new Map(); + for (const [index, entry] of parsed.tests.entries()) { + if (!entry || typeof entry !== 'object' || typeof entry.name !== 'string' || !entry.name + || typeof entry.passed !== 'boolean') { + manualProblems.push(`${rel}: attempt ${index + 1}: malformed collector entry (name/passed required)`); + continue; + } + const key = `${entry.suite ?? ''}\0${entry.name}`; + const occurrence = (seen.get(key) ?? 0) + 1; + seen.set(key, occurrence); + if (Object.hasOwn(entry, 'manual_review') && occurrence !== 1) { + manualProblems.push(`${rel}: attempt ${index + 1}: manual review is only valid on the first case attempt`); + } + if (Object.hasOwn(entry, 'manual_review')) { + const previous = manualClaims.get(key); + if (previous && previous !== rel) manualProblems.push(`${rel}: duplicate manual-review claim for ${entry.name} (also in ${previous})`); + else manualClaims.set(key, rel); + } + const problem = manualReviewProblem(entry, rootDir); + if (problem) manualProblems.push(`${rel}: attempt ${index + 1}: ${problem}`); + } + collectors.push(parsed); + const counts = collectorOutcomeCounts([parsed]); + files.push({ file: rel, tier: parsed.tier ?? 'unknown', shard: parsed.shard ?? '-', + cost: parsed.total_cost_usd ?? 0, flaky: parsed.flaky_retries?.length ?? 0, + total: counts.passed + counts.failed + counts.manual_accepted, ...counts }); + for (const f of parsed.flaky_retries ?? []) flaky.push({ ...f, file: rel }); + } catch (error) { + manualProblems.push(`${rel}: malformed collector JSON (${error instanceof Error ? error.message : String(error)})`); + } + } + const evidence = collectorOutcomeCounts(collectors); + console.log(`[test:paid] collector final outcomes: ${evidence.executed} executed, ${evidence.reused} reused; ${evidence.passed} passed, ${evidence.failed} failed, ${evidence.manual_accepted} manual accepted (unscored; no score-cache credit) (${evidence.attempts} attempt records from ${collectors.length} collectors; every record counts)`); + if (flaky.length > 0) { + console.log(`[test:paid] report: ⚠ ${flaky.length} cases with multiple attempts this run: (paid evals never retry; each record counts)`); + for (const f of flaky) console.log(` ⚠ ${f.name} (x${f.attempts}) — ${f.file}`); + } + const allSkipped = results.flatMap((r) => r.outcomes.filter(isAllSkippedPass)); + if (allSkipped.length > 0) { + console.log(`[test:paid] report: ⚠ ${allSkipped.length} shard(s) passed with EVERY test skipped — they verified nothing:`); + for (const outcome of allSkipped) { + console.log(` ⚠ ${outcome.files.join(' ')} (${outcome.executedTests} skipped — external service missing or tier mismatch)`); + } + } + if (manualProblems.length) verdict.problems.push(...manualProblems); + if (evidence.failed > 0) verdict.problems.push(`${evidence.failed} unapproved final collector failure(s)`); + if (files.reduce((sum, file) => sum + file.total, 0) !== evidence.passed + evidence.failed + evidence.manual_accepted + || files.reduce((sum, file) => sum + file.executed + file.reused, 0) !== evidence.executed + evidence.reused) { + verdict.problems.push('Collector summary totals are inconsistent'); + } + + // Panel verdicts: one function, computed here only. + const panels = panelReports(manifest, results, primary); + for (const panel of panels.filter(p => p.failsLane)) { + verdict.problems.push(`PANEL ${panel.case} ${panel.status}${panel.redClass === 'INFRA' ? ' (INFRA)' : ''} ${panel.passed}/${panel.panel.n} (${panel.marks}): ${panel.reason}`); + } + const laterPanels = attempts.slice(1).flatMap(attempt => panelReports(manifest, artifacts.map(a => a.result), attempt) + .filter(panel => panel.trials.length > 0)); + + // Quarantine policy checks on census runs: the per-tier cap and entry expiry. + if (manifest.evalsAll) { + const tierIds = Object.keys(E2E_TIERS).filter(id => E2E_TIERS[id] === manifest.tier); + const quarantined = Object.keys(CASE_QUARANTINE).filter(id => E2E_TIERS[id] === manifest.tier); + if (quarantined.length > EVAL_POLICY.quarantine.capFraction * tierIds.length) { + verdict.problems.push(`QUARANTINE over cap: ${quarantined.length} of ${tierIds.length} ${manifest.tier} cases (cap ${Math.round(EVAL_POLICY.quarantine.capFraction * 100)}%)`); + } + const expiryMs = EVAL_POLICY.quarantine.expiryWeeklyRuns * 7 * 24 * 60 * 60 * 1000; + for (const id of quarantined) { + const entered = Date.parse(CASE_QUARANTINE[id]!.enteredAt); + if (!Number.isFinite(entered) || Date.now() - entered > expiryMs) { + verdict.problems.push(`QUARANTINE expired: ${id} (entered ${CASE_QUARANTINE[id]!.enteredAt}; entries expire after ${EVAL_POLICY.quarantine.expiryWeeklyRuns} weekly runs)`); + } + } + } + + // History: one trial-outcomes line per isolated trial and per JUnit rule/judge case. + const runId = env.GITHUB_RUN_ID; + const sha = env.GITHUB_SHA; + const history: TrialOutcomeRecord[] = []; + const common = (attempt: number) => ({ schema: TRIAL_OUTCOME_SCHEMA, tier: manifest.tier, attempt, policy_version: EVAL_POLICY.version, + ...(runId ? { run_id: runId } : {}), ...(sha ? { sha } : {}), lane, recorded_at: new Date().toISOString() }); + for (const { result } of artifacts) { + const attempt = result.attempt ?? 1; + for (const outcome of result.outcomes) { + const t = outcome.trial; + if (!t || t.outcome === null) continue; + history.push({ ...common(attempt), case: t.case, file: shardFile(outcome.files[0]!), kind: t.kind, trial: t.trial, panel: t.panel, + outcome: t.outcome, ...(t.outcome === 'failed' ? { failure_class: t.failure_class ?? 'assertion' } : {}), + ...(t.exit_reason ? { exit_reason: t.exit_reason } : {}), ...(t.error ? { error: t.error } : {}), + duration_ms: t.duration_ms, cost_usd: t.cost_usd, ...(t.model ? { model: t.model } : {}), + ...(outcome.reused ? { input_identity: outcome.reused.inputKey } : {}), + quarantined: t.quarantined, execution: outcome.reused ? 'reused' : 'executed', source: 'shard' } as TrialOutcomeRecord); + } + } + const ruleCases: Array<{ id: string; kind: EvalCaseKind; outcome: TrialOutcome; line?: string }> = []; + const junitFailedShards = new Set(); + let unattributed = 0; + const cliVersion = env.GSTACK_CLAUDE_CLI_VERSION; + for (const { root, result } of artifacts.filter(a => (a.result.attempt ?? 1) === primary)) { + for (const outcome of result.outcomes) { + const key = normalizeRelativePath(outcome.files[0] ?? ''); + if (outcome.trial || outcome.files.length !== 1) continue; + let xml = ''; + try { xml = fs.readFileSync(path.join(root, 'shards', shardSlug([key]), 'junit.xml'), 'utf8'); } catch { continue; } + const records = recordsByShard.get(key) ?? []; + for (const tc of parseJUnitCases(xml)) { + const id = caseIdForTestName(tc.name); + if (id === null) { unattributed++; continue; } + const kind = (E2E_KINDS[id] ?? 'rule') as EvalCaseKind; + const mine = records.filter((r: any) => r?.name === id || r?.case_id === id); + const failedRecord = mine.find((r: any) => r.passed === false); + const failureClass: TrialFailureClass | undefined = tc.outcome !== 'failed' ? undefined + : tc.failureType === 'TimeoutError' ? 'timeout' : failedRecord ? failureClassOf(failedRecord) : 'assertion'; + const error = sanitizeTrialError(failedRecord?.error ?? tc.message); + if (tc.outcome === 'failed') junitFailedShards.add(key); + ruleCases.push({ id, kind, outcome: tc.outcome, + ...(tc.outcome === 'failed' ? { line: `✗ ${id} ${kind} FAIL ${failureClass}${failedRecord?.exit_reason === 'timeout' && failedRecord?.timeout_at_turn !== undefined ? ` at turn ${failedRecord.timeout_at_turn}` : ''}${error ? ` — ${error}` : ''} [slice ${result.sliceIndex}, attempt ${primary}] rerun: ${rerunCommand(manifest.tier, id, 1)}` } : {}) }); + history.push({ ...common(primary), case: id, file: shardFile(key), kind, trial: 1, panel: { n: 1, k: 1 }, outcome: tc.outcome, + ...(failureClass ? { failure_class: failureClass } : {}), ...(failedRecord?.exit_reason ? { exit_reason: String(failedRecord.exit_reason) } : {}), + ...(error && tc.outcome === 'failed' ? { error } : {}), duration_ms: tc.timeMs, + cost_usd: Math.round(mine.reduce((sum: number, r: any) => sum + (Number(r.cost_usd) || 0), 0) * 100) / 100, + ...(typeof mine[0]?.model === 'string' ? { model: mine[0].model } : {}), ...(cliVersion ? { cli_version: cliVersion } : {}), + quarantined: false, execution: outcome.reused ? 'reused' : 'executed', source: 'junit' } as TrialOutcomeRecord); + } + } + } + fs.writeFileSync(trialOutcomesPath, formatTrialOutcomes(history)); + + // Headline and failure block (A4): one formatter for the log, the PR comment and the weekly issue. + const ruleShardFailures = manifest.entries.filter(entry => entry.status === 'planned' && !entry.trial).flatMap(entry => { + const got = results.flatMap(r => r.outcomes.map(o => ({ o, slice: r.sliceIndex }))).find(({ o }) => normalizeRelativePath(o.files[0] ?? '') === normalizeRelativePath(entry.file)); + if (got && got.o.status === 'passed') return []; + if (got && junitFailedShards.has(normalizeRelativePath(entry.file))) return []; + const id = shardCaseId(entry.file); + return [`✗ ${entry.file} rule shard ${got ? got.o.status : 'NOT REPORTED'}${got?.o.runnerError ? ` — ${sanitizeTrialError(got.o.runnerError)}` : ''} [slice ${entry.slice}, attempt ${primary}]${id ? ` rerun: ${rerunCommand(manifest.tier, id, 1)}` : ''}`]; + }); + const behaviorPanels = panels.filter(p => !p.quarantined && p.kind === 'behavior'); + const lanePanels = panels.filter(p => !p.quarantined); + const count = (kind: EvalCaseKind) => ({ + passed: ruleCases.filter(c => c.kind === kind && c.outcome === 'passed').length + + lanePanels.filter(p => p.kind === kind && p.status === 'PASS').length, + total: ruleCases.filter(c => c.kind === kind).length + lanePanels.filter(p => p.kind === kind).length, + }); + const primaryTrials = results.flatMap(r => r.outcomes.map(o => o.trial)).filter((t): t is ShardTrialRecord => !!t); + const wall = results.filter(r => Number.isSafeInteger(r.startedAt) && Number.isSafeInteger(r.finishedAt)); + const failureLines = [ + ...ruleShardFailures, + ...ruleCases.filter(c => c.line).map(c => c.line!), + ...panels.filter(p => p.status !== 'PASS' || p.split).map(p => formatPanelLine(p, manifest.tier)), + ]; + const red = verdict.problems.length > 0; + const headline: ReportHeadline = { + lane, verdict: red ? 'RED' : 'GREEN', attempt: primary, + counts: { + rule: count('rule'), + behavior: { ...count('behavior'), split: behaviorPanels.filter(p => p.split).length }, + judge: count('judge'), + quarantined: { total: panels.filter(p => p.quarantined).length, failingLane: panels.filter(p => p.quarantined && p.failsLane).length }, + skipped: allSkipped.length + panels.filter(p => p.status === 'SKIPPED').length + ruleCases.filter(c => c.outcome === 'skipped').length, + infra: primaryTrials.filter(t => t.failure_class === 'infra').length + + results.flatMap(r => r.outcomes).filter(o => !o.trial && o.runnerError !== undefined).length, + incomplete: panels.filter(p => p.status === 'INCOMPLETE').length, + unattributed, + }, + actionRequired: verdict.problems.length, + wallMs: wall.length ? Math.max(...wall.map(r => r.finishedAt!)) - Math.min(...wall.map(r => r.startedAt!)) : null, + costUsd: Math.round(costUsd * 100) / 100, + redispatchEligible: red && infraOnly(verdict.problems), + }; + const headlineLines = formatHeadline(headline); + for (const line of headlineLines) console.log(line); + if (failureLines.length) { + console.log('[test:paid] failures and split verdicts:'); + for (const line of failureLines) console.log(` ${line}`); + } + for (const panel of laterPanels) console.log(` attempt ${panel.attempt} (re-run; reported, never replacing attempt ${primary}): ${formatPanelLine(panel, manifest.tier)}`); + const fence = (lines: string[]) => ['```', ...lines.map(line => line.replace(/```/g, "'''")), '```']; + fs.writeFileSync(summaryMdPath, [ + ...fence(headlineLines), + ...(failureLines.length ? ['', '**Failures and split verdicts**', '', ...fence(failureLines)] : []), + ...(verdict.problems.length ? ['', `**ACTION REQUIRED (${verdict.problems.length})**`, '', ...fence(verdict.problems.map(p => sanitizeTrialError(p) ?? p))] : []), + ].join('\n') + '\n'); + if (!manualProblems.length) fs.writeFileSync(summaryPath, JSON.stringify({ version: 2, files, totals: { + ...evidence, total: evidence.passed + evidence.failed + evidence.manual_accepted, + flaky: files.reduce((sum, file) => sum + file.flaky, 0), + }, verdict: headline, headline: headlineLines, + panels: panels.map(p => ({ case: p.case, kind: p.kind, status: p.status, passed: p.passed, n: p.panel.n, k: p.panel.k, + marks: p.marks, split: p.split, quarantined: p.quarantined, failsLane: p.failsLane, redClass: p.redClass, reason: p.reason, + trials: p.trials.map(t => ({ trial: t.trial, outcome: t.outcome, ...(t.failure_class ? { failure_class: t.failure_class } : {}), + ...(t.exit_reason ? { exit_reason: t.exit_reason } : {}), ...(t.error ? { error: t.error } : {}) })) })), + failures: failureLines.map(line => sanitizeTrialError(line) ?? line) }, null, 2) + '\n'); + if (verdict.problems.length) { + console.error(`[test:paid] report: ${verdict.problems.length} problem(s):`); + for (const problem of verdict.problems) console.error(` ✗ ${problem}`); + if (headline.redispatchEligible) console.error('[test:paid] report: INFRA-ONLY RED — one re-dispatch as a new run is allowed; report both runs'); + return 1; + } + console.log(evidence.manual_accepted + ? `[test:paid] report: every planned shard accounted; ${evidence.manual_accepted} manual acceptance(s), no automated-score credit` + : '[test:paid] report: every planned shard accounted and passed'); + return 0; +} + type CliOptions = { tier: PaidTier; profile: PaidProfile; @@ -1838,6 +2670,8 @@ type CliOptions = { reportDir: string | null; /** Report mode: merge executed shard wall times into the duration seed. */ writeDurations: boolean; + /** Planner: the workflow matrix cap, for the capacity preflight's wave count. */ + maxParallel: number | null; }; function parsePositiveInt(value: string | undefined, flag: string): number { @@ -1892,6 +2726,7 @@ export function parseCliOptions(argv: string[], env: NodeJS.ProcessEnv = process sliceIndex: null, reportDir: null, writeDurations: false, + maxParallel: null, }; for (let index = 0; index < argv.length; index += 1) { @@ -1929,6 +2764,7 @@ export function parseCliOptions(argv: string[], env: NodeJS.ProcessEnv = process options.reportDir = value; continue; } if (arg === '--write-durations') { options.writeDurations = true; continue; } + if (arg === '--max-parallel') { options.maxParallel = parsePositiveInt(argv[index += 1], '--max-parallel'); continue; } throw new Error(`Unknown argument: ${arg}`); } if (options.writeDurations && !options.reportDir) throw new Error('--write-durations requires --report'); @@ -1965,125 +2801,13 @@ async function main(): Promise { + `${excludedCount} excluded (${manifest.selectionReason})`, ); for (const line of formatSlicePlan(manifest)) console.log(line); + for (const line of formatCapacityPreflight(manifest, options.maxParallel ?? undefined)) console.log(line); return 0; } // ── Report mode: reconcile slice artifacts against the manifest. Fail-closed: // a slice whose artifact never landed is a FAILURE, not an absence. - const reportDir = options.reportDir; - if (reportDir) { - const summaryPath = path.join(reportDir, 'collector-outcomes.json'); - fs.rmSync(summaryPath, { force: true }); - const manifest = parseRunManifest(fs.readFileSync(path.join(reportDir, 'manifest.json'), 'utf-8')); - const results: SliceResult[] = fs.readdirSync(reportDir) - .filter((name) => /^slice-\d+\.json$/.test(name)) - .map((name) => JSON.parse(fs.readFileSync(path.join(reportDir, name), 'utf-8')) as SliceResult); - const verdict = verifySliceResults(manifest, results); - const planned = manifest.entries.filter((e) => e.status === 'planned').length; - console.log(`[test:paid] report: ${results.length}/${manifest.sliceCount} slices, ${planned} planned shards, tier=${manifest.tier}`); - for (const line of formatProfileCoverage(manifest)) console.log(line); - for (const result of results.sort((a, b) => a.sliceIndex - b.sliceIndex)) { - for (const outcome of result.outcomes) { - const shown = outcome.reused ? `reused (run ${outcome.reused.runId})` : outcome.status; - console.log(` slice ${result.sliceIndex} ${shown.padEnd(15)} ${String(Math.round(outcome.elapsedMs / 1000)).padStart(5)}s ${outcome.files.join(' ')}`); - } - } - if (options.writeDurations) { - const durations = mergePaidTestDurations(loadPaidTestDurations(ROOT, manifest.tier), results); - writePaidTestDurations(manifest.tier, durations); - console.log(`[test:paid] wrote ${Object.keys(durations).length} ${manifest.tier} durations to ${PAID_TEST_DURATIONS_FILE}`); - } - // Historical flaky_retries includes every case with multiple attempts, - // whether its final result passed or failed. Report attempts separately - // from the shard verdict; reconciliation above still controls gating. - // Source: the finalized eval-store JSONs inside the slice artifacts. - const flaky: Array<{ name: string; attempts: number; file: string }> = []; - const collectors: Parameters[0] = []; - const files: Array<{ file: string; tier: string; shard: string | number; cost: number; - flaky: number; total: number; executed: number; reused: number; passed: number; - failed: number; manual_accepted: number; attempts: number }> = []; - const manualProblems: string[] = []; - const manualClaims = new Map(); - for (const name of fs.readdirSync(reportDir, { recursive: true }) as string[]) { - if (!isFinalizedEvalResultFile(name)) continue; - try { - const parsed = JSON.parse(fs.readFileSync(path.join(reportDir, name), 'utf-8')); - if (!Array.isArray(parsed.tests)) { - if (Object.hasOwn(parsed, 'tests') || parsed.total_tests !== undefined || parsed.manual_review !== undefined) { - manualProblems.push(`${name}: malformed collector tests[]`); - } - continue; - } - const seen = new Map(); - for (const [index, entry] of parsed.tests.entries()) { - if (!entry || typeof entry !== 'object' || typeof entry.name !== 'string' || !entry.name - || typeof entry.passed !== 'boolean') { - manualProblems.push(`${name}: attempt ${index + 1}: malformed collector entry (name/passed required)`); - continue; - } - const key = `${entry.suite ?? ''}\0${entry.name}`; - const occurrence = (seen.get(key) ?? 0) + 1; - seen.set(key, occurrence); - if (Object.hasOwn(entry, 'manual_review') && occurrence !== 1) { - manualProblems.push(`${name}: attempt ${index + 1}: manual review is only valid on the first case attempt`); - } - if (Object.hasOwn(entry, 'manual_review')) { - const previous = manualClaims.get(key); - if (previous && previous !== name) manualProblems.push(`${name}: duplicate manual-review claim for ${entry.name} (also in ${previous})`); - else manualClaims.set(key, name); - } - const problem = manualReviewProblem(entry, ROOT); - if (problem) manualProblems.push(`${name}: attempt ${index + 1}: ${problem}`); - } - collectors.push(parsed); - const counts = collectorOutcomeCounts([parsed]); - files.push({ file: name, tier: parsed.tier ?? 'unknown', shard: parsed.shard ?? '-', - cost: parsed.total_cost_usd ?? 0, flaky: parsed.flaky_retries?.length ?? 0, - total: counts.passed + counts.failed + counts.manual_accepted, ...counts }); - for (const f of parsed.flaky_retries ?? []) flaky.push({ ...f, file: name }); - } catch (error) { - manualProblems.push(`${name}: malformed collector JSON (${error instanceof Error ? error.message : String(error)})`); - } - } - const evidence = collectorOutcomeCounts(collectors); - console.log(`[test:paid] collector final outcomes: ${evidence.executed} executed, ${evidence.reused} reused; ${evidence.passed} passed, ${evidence.failed} failed, ${evidence.manual_accepted} manual accepted (unscored; no score-cache credit) (${evidence.attempts} attempt records from ${collectors.length} collectors)`); - if (flaky.length > 0) { - console.log(`[test:paid] report: ⚠ ${flaky.length} cases with multiple attempts this run:`); - for (const f of flaky) console.log(` ⚠ ${f.name} (x${f.attempts}) — ${f.file}`); - } - - // Census honesty: a 'passed' shard whose every test skipped verified - // nothing (external-service binary absent on the runner). Not a failure — - // service availability is host state, not a repo regression — but the - // report must say so, or the weekly lane reads codex/gemini as covered - // on runners that never install them. - const allSkipped = results.flatMap((r) => r.outcomes.filter(isAllSkippedPass)); - if (allSkipped.length > 0) { - console.log(`[test:paid] report: ⚠ ${allSkipped.length} shard(s) passed with EVERY test skipped — they verified nothing:`); - for (const outcome of allSkipped) { - console.log(` ⚠ ${outcome.files.join(' ')} (${outcome.executedTests} skipped — external service missing or tier mismatch)`); - } - } - if (manualProblems.length) verdict.problems.push(...manualProblems); - if (evidence.failed > 0) verdict.problems.push(`${evidence.failed} unapproved final collector failure(s)`); - if (files.reduce((sum, file) => sum + file.total, 0) !== evidence.passed + evidence.failed + evidence.manual_accepted - || files.reduce((sum, file) => sum + file.executed + file.reused, 0) !== evidence.executed + evidence.reused) { - verdict.problems.push('Collector summary totals are inconsistent'); - } - if (!manualProblems.length) fs.writeFileSync(summaryPath, JSON.stringify({ version: 1, files, totals: { - ...evidence, total: evidence.passed + evidence.failed + evidence.manual_accepted, - flaky: files.reduce((sum, file) => sum + file.flaky, 0), - } }, null, 2) + '\n'); - if (verdict.problems.length) { - console.error(`[test:paid] report: ${verdict.problems.length} problem(s):`); - for (const problem of verdict.problems) console.error(` ✗ ${problem}`); - return 1; - } - console.log(evidence.manual_accepted - ? `[test:paid] report: every planned shard accounted; ${evidence.manual_accepted} manual acceptance(s), no automated-score credit` - : '[test:paid] report: every planned shard accounted and passed'); - return 0; - } + if (options.reportDir) return runPaidReport(options.reportDir, { writeDurations: options.writeDurations }); const discovered = collectPaidTestFiles(); if (discovered.length === 0) throw new Error('No paid test files were discovered.'); @@ -2119,28 +2843,36 @@ async function main(): Promise { } const evalDirBase = process.env.GSTACK_EVAL_DIR || getProjectEvalDir(); + const trials = Object.fromEntries(mine.filter(entry => entry.trial).map(entry => [normalizeRelativePath(entry.file), entry.trial!])); + const exclusionPatterns = Object.fromEntries(mine.filter(entry => entry.excludeCases).map(entry => [entry.file, + manifest.prCoverage?.mode === 'pr' ? prProfileTestNamePattern(entry.file, manifest.selection!, entry.excludeCases) + : excludedCasesNamePattern(entry.excludeCases!)])); + const startedAt = Date.now(); let summary: RunSummary; if (shards.length === 0) { summary = summarize([]); } else { preflightAnthropicApi(process.env); summary = await runPaidShards(shards, { + trials, + casePatterns: exclusionPatterns, timeoutMs: options.timeoutExplicit ? options.timeoutMs : undefined, jobs: options.jobs, withinShardConcurrency: options.withinShardConcurrency, registeredBudgets: Object.fromEntries(mine.filter(entry => entry.budget).map(entry => [normalizeRelativePath(entry.file), entry.budget!])), ...(manifest.prCoverage?.mode === 'pr' ? { - expectedCases: Object.fromEntries(mine.map(entry => [entry.file, expectedPrCaseCount(entry.file, manifest.selection!)])), - casePatterns: Object.fromEntries(mine.map(entry => [entry.file, prProfileTestNamePattern(entry.file, manifest.selection!)])), - expectedCaseIds: Object.fromEntries(mine.map(entry => [entry.file, prProfileShardIds(entry.file, manifest.selection!)])), + expectedCases: Object.fromEntries(mine.map(entry => [entry.file, expectedPrCaseCount(entry.file, manifest.selection!, entry.excludeCases)])), + casePatterns: Object.fromEntries(mine.map(entry => [entry.file, prProfileTestNamePattern(entry.file, manifest.selection!, entry.excludeCases)])), + expectedCaseIds: Object.fromEntries(mine.map(entry => [entry.file, prProfileShardIds(entry.file, manifest.selection!, entry.excludeCases)])), reuseFor: e2eReuseLaneProblem(process.env, manifest.prCoverage.mode) !== null ? undefined : (files, env, budget) => { const key = files[0]!; const file = shardFile(key); if (files.length !== 1 || !/^test\/skill-e2e-/.test(file)) return null; const { registered, known } = fileCaseRegistration(file, fs.readFileSync(path.join(ROOT, file), 'utf8')); - return prepareE2EShardReuse({ root: ROOT, key, file, caseIds: prProfileShardIds(key, manifest.selection!), + const exclude = mine.find(entry => entry.file === key)?.excludeCases; + return prepareE2EShardReuse({ root: ROOT, key, file, caseIds: prProfileShardIds(key, manifest.selection!, exclude), registeredIds: registered, registrationKnown: known, - casePattern: prProfileTestNamePattern(key, manifest.selection!), expectedCases: expectedPrCaseCount(key, manifest.selection!), + casePattern: prProfileTestNamePattern(key, manifest.selection!, exclude), expectedCases: expectedPrCaseCount(key, manifest.selection!, exclude), retries: retriesForFiles(files), timeoutMs: budget.timeoutMs, withinShardConcurrency: options.withinShardConcurrency, tier: manifest.tier, profile, env }); }, @@ -2161,8 +2893,9 @@ async function main(): Promise { evalDirBase, }); } - const guarded = applyHollowShardGuard(summary.outcomes, { evalsAll: manifest.evalsAll, requireExecuted: manifest.prCoverage?.mode === 'pr' }); + const guarded = guardTrialRecords(applyHollowShardGuard(summary.outcomes, { evalsAll: manifest.evalsAll, requireExecuted: manifest.prCoverage?.mode === 'pr' })); summary = summarize(guarded); + const attempt = Number(process.env.GITHUB_RUN_ATTEMPT); const sliceResult: SliceResult = { version: 1, tier: manifest.tier, @@ -2171,15 +2904,23 @@ async function main(): Promise { sliceIndex: options.sliceIndex, sliceCount: manifest.sliceCount, ...(options.timeoutExplicit ? { timeoutOverrideMs: options.timeoutMs } : {}), - outcomes: guarded.map(({ files, status, exitCode, elapsedMs, executedTests, skippedTests, budget, reused }) => - ({ files, status, exitCode, elapsedMs, executedTests, skippedTests, ...(budget ? { budget } : {}), ...(reused ? { reused } : {}) })), + attempt: Number.isSafeInteger(attempt) && attempt > 0 ? attempt : 1, + startedAt, + finishedAt: Date.now(), + outcomes: guarded.map(({ files, status, exitCode, elapsedMs, executedTests, skippedTests, budget, reused, runnerError, trial }) => + ({ files, status, exitCode, elapsedMs, executedTests, skippedTests, ...(budget ? { budget } : {}), ...(reused ? { reused } : {}), + ...(runnerError !== undefined ? { runnerError } : {}), ...(trial ? { trial } : {}) })), }; fs.mkdirSync(evalDirBase, { recursive: true }); const sliceResultPath = path.join(evalDirBase, `slice-${options.sliceIndex}.json`); fs.writeFileSync(sliceResultPath, `${JSON.stringify(sliceResult, null, 2)}\n`); console.log(`[test:paid] slice result: ${sliceResultPath}`); for (const line of formatSummary(summary)) console.log(line); - return summaryExitCode(summary); + for (const outcome of guarded.filter(outcome => outcome.trial)) { + const t = outcome.trial!; + console.log(` trial ${t.case} t${t.trial}/${t.panel.n}: ${t.outcome ?? `NO RECORD (${t.harness})`}${t.failure_class ? ` [${t.failure_class}]` : ''}`); + } + return sliceExitCode(guarded); } if (options.listOnly && options.sliceBudgetMs !== null) { diff --git a/test/ci-paid-coordination.test.ts b/test/ci-paid-coordination.test.ts index 18c80adfc..7405037e9 100644 --- a/test/ci-paid-coordination.test.ts +++ b/test/ci-paid-coordination.test.ts @@ -248,7 +248,8 @@ describe('dependency-free CI planner and report execution', () => { const red = run(['--report', reportDir], tier); expect(red.status).toBe(1); expect(red.stderr).toContain(`${failed.outcomes[0].files[0]}: failed`); - expect(red.stdout).toContain('3 executed, 0 reused; 1 passed, 2 failed, 0 manual accepted (unscored; no score-cache credit) (6 attempt records from 1 collectors)'); + // Paid evals never retry: every record counts, a later pass never hides an earlier failure. + expect(red.stdout).toContain('6 executed, 0 reused; 2 passed, 4 failed, 0 manual accepted (unscored; no score-cache credit) (6 attempt records from 1 collectors'); expect(red.stdout).toContain('3 cases with multiple attempts this run:'); expect(red.stdout).not.toMatch(/passed only on retry|not blocking/); diff --git a/test/paid-pr-profile.test.ts b/test/paid-pr-profile.test.ts index 34609e54f..8e46944df 100644 --- a/test/paid-pr-profile.test.ts +++ b/test/paid-pr-profile.test.ts @@ -241,7 +241,7 @@ describe('PR profile paid-runner integration', () => { expect(guarded[0].status).toBe('passed-empty'); }); - test('report distinguishes retained/deferred coverage and final executed/reused outcomes from attempts', () => { + test('report distinguishes retained/deferred coverage and counts every executed/reused record', () => { const manifest = ceoManifest(); const lines = formatProfileCoverage(manifest).join('\n'); expect(lines).toContain('profile=pr mode=pr'); @@ -252,6 +252,7 @@ describe('PR profile paid-runner integration', () => { { name: 'retry', suite: 'judge', passed: true, execution: 'executed' }, { name: 'cached', suite: 'judge', passed: true, execution: 'reused' }, { name: 'failed', suite: 'native', passed: false }, - ] }])).toEqual({ executed: 2, reused: 1, passed: 2, failed: 1, manual_accepted: 0, attempts: 4 }); + // Paid evals never retry: a later pass never replaces an earlier failed record. + ] }])).toEqual({ executed: 3, reused: 1, passed: 2, failed: 2, manual_accepted: 0, attempts: 4 }); }); }); diff --git a/test/paid-report-fail-open.test.ts b/test/paid-report-fail-open.test.ts index cc734f998..a286c737c 100644 --- a/test/paid-report-fail-open.test.ts +++ b/test/paid-report-fail-open.test.ts @@ -108,3 +108,110 @@ describe('rule shards stay fail-closed through --report', () => { expect(r.out).toContain(`${RULE_B} planned for slice 2 but reported by slice 1`); }); }); + +describe('behavior and quarantined panels through --report', () => { + const FILE = 'test/skill-e2e-review.test.ts'; + const ID = 'review-design-lite'; + const key = (trial: number) => `${FILE}#${ID}~t${trial}`; + const plan = (quarantined = false) => manifest([ + ...[1, 2, 3].map(trial => ({ file: key(trial), slice: trial, status: 'planned' as const, + trial: { kind: 'behavior' as const, panel: { n: 3, k: 2 }, quarantined } })), + { file: RULE_A, slice: 4, status: 'planned' }, + ], 4); + type TrialResult = 'passed' | 'failed' | 'contract' | 'missing' | 'harness'; + const trialOutcome = (trial: number, result: TrialResult, quarantined = false): Outcome | null => { + if (result === 'missing') return null; + const record = { case: ID, trial, kind: 'behavior' as const, panel: { n: 3, k: 2 }, quarantined, cost_usd: 0, duration_ms: 1_000 }; + if (result === 'harness') return passed(key(trial), { status: 'never-started', exitCode: null, executedTests: null, skippedTests: null, + trial: { ...record, outcome: null, harness: 'never started' } }); + if (result === 'passed') return passed(key(trial), { trial: { ...record, outcome: 'passed' } }); + return passed(key(trial), { status: 'failed', exitCode: 1, + trial: { ...record, outcome: 'failed', failure_class: result === 'contract' ? 'contract' : 'timeout', + exit_reason: 'timeout', timeout_at_turn: 14, error: result === 'contract' ? 'handoff missing' : 'no posture match' } }); + }; + const run = (results: TrialResult[], quarantined = false, dropSlice?: number) => report(plan(quarantined), [1, 2, 3, 4] + .filter(index => index !== dropSlice) + .map(index => slice(index, 4, index === 4 ? [passed(RULE_A)] + : [trialOutcome(index, results[index - 1]!, quarantined)].filter((o): o is Outcome => o !== null)))); + + test('behavior 3/3: green', () => { + const r = run(['passed', 'passed', 'passed']); + expect(r.status, r.out).toBe(0); + expect(r.out).toContain('VERDICT GREEN'); + }); + + test('behavior 2/3: green, the failed trial shown with its cause', () => { + const r = run(['passed', 'failed', 'passed']); + expect(r.status, r.out).toBe(0); + expect(r.out).toContain(`⚠ ${ID} behavior PASS 2/3 (✓✗✓)`); + expect(r.out).toContain('t2: timeout at turn 14'); + const summary = JSON.parse(fs.readFileSync(path.join(r.dir, 'collector-outcomes.json'), 'utf8')); + expect(summary.version).toBe(2); + expect(summary.panels[0]).toMatchObject({ case: ID, status: 'PASS', split: true, failsLane: false }); + const history = fs.readFileSync(path.join(r.dir, 'trial-outcomes.jsonl'), 'utf8').trim().split('\n').map(line => JSON.parse(line)); + expect(history.map(h => [h.trial, h.outcome])).toEqual([[1, 'passed'], [2, 'failed'], [3, 'passed']]); + }); + + test('behavior 1/3: red', () => { + const r = run(['passed', 'failed', 'failed']); + expect(r.status).toBe(1); + expect(r.out).toContain(`PANEL ${ID} FAIL 1/3`); + }); + + test('a missing trial record: INCOMPLETE, red', () => { + const r = run(['passed', 'missing', 'passed']); + expect(r.status).toBe(1); + expect(r.out).toContain(`PANEL ${ID} INCOMPLETE`); + }); + + test('a trial the harness never started: red, machine-classified for one re-dispatch', () => { + const r = run(['passed', 'harness', 'passed']); + expect(r.status).toBe(1); + expect(r.out).toContain('no trial record (never started)'); + expect(r.out).toContain('INFRA-ONLY RED'); + }); + + test('a contract trial at 2/3: red', () => { + const r = run(['passed', 'passed', 'contract']); + expect(r.status).toBe(1); + expect(r.out).toContain(`PANEL ${ID} FAIL 2/3`); + expect(r.out).toContain('contract violation'); + expect(r.out).not.toContain('INFRA-ONLY RED'); + }); + + test('quarantined 1/3: reported, does not fail the lane', () => { + const r = run(['passed', 'failed', 'failed'], true); + expect(r.status, r.out).toBe(0); + expect(r.out).toContain(`◌ ${ID} behavior (quarantined) FAIL 1/3`); + }); + + test('quarantined 0/3: hard break, red', () => { + const r = run(['failed', 'failed', 'failed'], true); + expect(r.status).toBe(1); + expect(r.out).toContain('quarantined hard break'); + }); + + test('quarantined contract violation: red', () => { + const r = run(['passed', 'passed', 'contract'], true); + expect(r.status).toBe(1); + }); + + test('a missing trial slice: red', () => { + const r = run(['passed', 'passed', 'passed'], false, 2); + expect(r.status).toBe(1); + expect(r.out).toContain('slice 2/4 reported NO result'); + expect(r.out).toContain(`PANEL ${ID} INCOMPLETE`); + }); + + test('a later run attempt never replaces the first attempt verdict', () => { + const r = run(['passed', 'failed', 'failed']); + expect(r.status).toBe(1); + const retry = slice(3, 4, [trialOutcome(3, 'passed')!]); + fs.mkdirSync(path.join(r.dir, 'paid-slice-3-a2'), { recursive: true }); + fs.writeFileSync(path.join(r.dir, 'paid-slice-3-a2', 'slice-3.json'), JSON.stringify({ ...retry, attempt: 2 })); + const again = spawnSync(process.execPath, [path.join(ROOT, 'scripts/test-paid-shards.ts'), '--tier', 'periodic', '--report', r.dir], + { cwd: ROOT, encoding: 'utf8', timeout: 30_000 }); + expect(again.status).toBe(1); + expect(again.stdout).toContain('attempt 1 (later attempts 2 reported, never replacing it)'); + }); +}); diff --git a/test/paid-run-manifest.test.ts b/test/paid-run-manifest.test.ts index b91da2547..74090aace 100644 --- a/test/paid-run-manifest.test.ts +++ b/test/paid-run-manifest.test.ts @@ -37,11 +37,16 @@ import { summarize, summaryExitCode, verifySliceResults, + expandTrialShards, + formatCapacityPreflight, + shardSlug, type PaidRunManifest, type ShardOutcome, type SliceResult, } from '../scripts/test-paid-shards'; +import { E2E_KINDS } from './helpers/touchfiles-data'; + const ROOT = path.resolve(__dirname, '..'); const outcome = (over: Partial): ShardOutcome => ({ @@ -508,3 +513,98 @@ describe('retry parity', () => { expect(buildPaidShardArgs(['x'], 1000, 4).join(' ')).toContain('--retry 0'); }); }); + +describe('trial planner (behavior and quarantined panels)', () => { + const REVIEW = 'test/skill-e2e-review.test.ts'; + const budgetPlan = (tier: 'gate' | 'periodic', kinds: Record, quarantine: Record = {}) => + buildRunManifest({ tier, sliceBudgetMs: 540_000, jobs: 2, evalsAll: true, env: { EVALS_ALL: '1' }, + kinds: { ...E2E_KINDS, ...kinds }, quarantine }); + + test('a behavior case becomes three trial shards on three different slices; its file shard runs the rest', () => { + const manifest = budgetPlan('gate', { 'review-sql-injection': 'behavior' }); + const trials = manifest.entries.filter(entry => entry.file.startsWith(`${REVIEW}#review-sql-injection~t`)); + expect(trials.map(entry => entry.file)).toEqual([1, 2, 3].map(n => `${REVIEW}#review-sql-injection~t${n}`)); + expect(trials.every(entry => entry.status === 'planned')).toBe(true); + expect(new Set(trials.map(entry => entry.slice)).size).toBe(3); + expect(trials[0]!.trial).toEqual({ kind: 'behavior', panel: { n: 3, k: 2 }, quarantined: false }); + const fileShard = manifest.entries.find(entry => entry.file === REVIEW)!; + expect(fileShard.excludeCases).toEqual(['review-sql-injection']); + const slugs = manifest.entries.map(entry => shardSlug([entry.file])); + expect(new Set(slugs).size).toBe(slugs.length); + expect(parseRunManifest(JSON.stringify(manifest))).toEqual(manifest); + }); + + test('a file whose only tier case is isolated drops its file shard', () => { + const manifest = budgetPlan('periodic', { 'review-design-lite': 'behavior' }); + expect(manifest.entries.some(entry => entry.file === REVIEW)).toBe(false); + expect(manifest.entries.filter(entry => entry.file.startsWith(`${REVIEW}#`)).map(entry => entry.file)) + .toEqual([1, 2, 3].map(n => `${REVIEW}#review-design-lite~t${n}`)); + }); + + test('a quarantined rule case runs a full panel with k = n', () => { + const manifest = budgetPlan('gate', {}, { 'review-enum-completeness': { reason: 'r' } }); + const trial = manifest.entries.find(entry => entry.file === `${REVIEW}#review-enum-completeness~t1`)!; + expect(trial.trial).toEqual({ kind: 'rule', panel: { n: 3, k: 3 }, quarantined: true }); + }); + + test('slice-count plans keep trials on different slices too', () => { + const manifest = buildRunManifest({ tier: 'gate', sliceCount: 5, evalsAll: true, env: { EVALS_ALL: '1' }, + kinds: { ...E2E_KINDS, 'review-sql-injection': 'behavior', 'review-enum-completeness': 'behavior' } }); + for (const id of ['review-sql-injection', 'review-enum-completeness']) { + const slices = manifest.entries.filter(entry => entry.file.startsWith(`${REVIEW}#${id}~t`)).map(entry => entry.slice); + expect(new Set(slices).size).toBe(3); + } + }); + + test('judges and unknown ids cannot be isolated; unknown registrations throw', () => { + expect(() => budgetPlan('gate', { 'review/SKILL.md workflow': 'behavior' })).toThrow(/Only live E2E cases/); + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'trial-unknown-')); + try { + fs.mkdirSync(path.join(dir, 'test')); + fs.writeFileSync(path.join(dir, 'test/skill-e2e-x.test.ts'), 'const name = pick(); runSkillTest({ testName: name });'); + expect(() => expandTrialShards(['test/skill-e2e-x.test.ts'], 'gate', dir, { + kinds: { x: 'behavior' }, touchfiles: { x: ['test/skill-e2e-x.test.ts'] }, tiers: { x: 'gate' }, + })).toThrow(/statically known case registration/); + } finally { fs.rmSync(dir, { recursive: true, force: true }); } + }); + + test('parse rejects partial panels, shared runners, forged plans and stray exclusions', () => { + const manifest = budgetPlan('gate', { 'review-sql-injection': 'behavior' }); + const trialFiles = [1, 2, 3].map(n => `${REVIEW}#review-sql-injection~t${n}`); + const mutate = (fn: (m: PaidRunManifest) => void) => { const m = structuredClone(manifest); fn(m); return JSON.stringify(m); }; + expect(() => parseRunManifest(mutate(m => { m.entries = m.entries.filter(e => e.file !== trialFiles[1]); }))) + .toThrow(/exactly its 3 trials/); + expect(() => parseRunManifest(mutate(m => { + const [a, b] = trialFiles.map(f => m.entries.find(e => e.file === f)!); + b!.slice = a!.slice; + }))).toThrow(/share a slice/); + expect(() => parseRunManifest(mutate(m => { m.entries.find(e => e.file === trialFiles[0])!.trial!.panel.k = 1; }))) + .toThrow(/fixed policy panel/); + expect(() => parseRunManifest(mutate(m => { m.entries.find(e => e.file === trialFiles[0])!.trial = undefined; }))) + .toThrow(/fixed policy panel/); + expect(() => parseRunManifest(mutate(m => { m.entries.find(e => e.file === REVIEW)!.excludeCases = ['review-design-lite']; }))) + .toThrow(/exclude only cases/); + expect(() => parseRunManifest(mutate(m => { m.entries.find(e => e.file === REVIEW)!.trial = m.entries.find(e => e.file === trialFiles[0])!.trial; }))) + .toThrow(/Only trial shards/); + }); + + test('capacity preflight names slices, shards, waves and the longest indivisible trial', () => { + const manifest = budgetPlan('gate', { 'review-sql-injection': 'behavior' }); + const lines = formatCapacityPreflight(manifest, 16).join('\n'); + expect(lines).toContain(`${manifest.sliceCount} slice(s)`); + expect(lines).toContain('3 trial shard(s)'); + expect(lines).toContain(`wave(s) at max-parallel 16: ${Math.ceil(manifest.sliceCount / 16)}`); + expect(lines).toMatch(/longest indivisible trial ~\d+\.\dm \(test\/skill-e2e-review\.test\.ts#review-sql-injection~t\d\)/); + }); + + test('durations: trials record their longest wall under the case key and seed their own estimate', () => { + const key = `${REVIEW}#review-sql-injection`; + const merged = mergePaidTestDurations({}, [{ version: 1, tier: 'gate', sliceIndex: 1, sliceCount: 1, outcomes: [1, 2, 3].map(n => ({ + files: [`${key}~t${n}`], status: 'passed' as const, exitCode: 0, elapsedMs: n * 60_000, executedTests: 1, skippedTests: 0, + })) }]); + expect(merged).toEqual({ [key]: 180_000 }); + const packed = packBySliceBudget([1, 2, 3].map(n => `${key}~t${n}`), 540_000, 2, merged); + expect(packed.slices).toHaveLength(3); + expect(Object.values(packed.estimates)).toEqual([180_000, 180_000, 180_000]); + }); +}); diff --git a/test/paid-shards.test.ts b/test/paid-shards.test.ts index e40239457..5268eef0d 100644 --- a/test/paid-shards.test.ts +++ b/test/paid-shards.test.ts @@ -47,6 +47,14 @@ import { selectPaidTestFiles, buildRunManifest, parseRunManifest, + classifyTrialShard, + sliceExitCode, + guardTrialRecords, + parseJUnitCases, + caseIdForTestName, + shardTrial, + excludedCasesNamePattern, + type CaseTrialPlan, type ShardOutcome, } from '../scripts/test-paid-shards'; @@ -578,3 +586,92 @@ describe('all-skipped pass census', () => { expect(reviewLine).not.toContain('SKIPPED'); }); }); + +describe('isolated trial shards: record, classification and slice exit', () => { + const plan: CaseTrialPlan = { kind: 'behavior', panel: { n: 3, k: 2 }, quarantined: false }; + const key = (n: number) => `test/skill-e2e-review.test.ts#review-sql-injection~t${n}`; + const base = { status: 'passed' as const, executedTests: 1, skippedTests: 0, elapsedMs: 5 }; + const none = { records: [], contract: null }; + + test('trial keys keep their case id, file and index', () => { + expect(shardCaseId(key(2))).toBe('review-sql-injection'); + expect(shardFile(key(2))).toBe('test/skill-e2e-review.test.ts'); + expect(shardTrial(key(2))).toBe(2); + expect(shardTrial('test/skill-e2e-review.test.ts#review-sql-injection')).toBeNull(); + expect(shardSlug([key(2)])).toBe('skill-e2e-review--review-sql-injection.t2'); + }); + + test('classification: verdicts versus harness problems', () => { + const c = (over: Partial, evidence: { records: any[]; contract: string | null } = none) => + classifyTrialShard({ ...base, ...over }, 'review-sql-injection', 1, plan, evidence); + expect(c({}).outcome).toBe('passed'); + expect(c({}, { records: [], contract: 'handoff missing' })).toMatchObject({ outcome: 'failed', failure_class: 'contract', error: 'handoff missing' }); + expect(c({ status: 'failed' }, { records: [{ passed: false, exit_reason: 'timeout', timeout_at_turn: 9, error: 'x' }], contract: null })) + .toMatchObject({ outcome: 'failed', failure_class: 'timeout', timeout_at_turn: 9 }); + expect(c({ status: 'failed' }, { records: [{ passed: false, error: 'expected 3' }], contract: null })).toMatchObject({ outcome: 'failed', failure_class: 'assertion' }); + expect(c({ status: 'timed-out', executedTests: null, skippedTests: null })).toMatchObject({ outcome: 'failed', failure_class: 'timeout' }); + expect(c({ status: 'failed', executedTests: null, skippedTests: null })).toMatchObject({ outcome: 'failed', failure_class: 'infra' }); + expect(c({ status: 'failed', executedTests: 0, skippedTests: 0 })).toMatchObject({ outcome: 'failed', failure_class: 'infra' }); + expect(c({ executedTests: 1, skippedTests: 1 })).toMatchObject({ outcome: 'skipped' }); + for (const over of [{ status: 'never-started' as const }, { status: 'passed-empty' as const }, { executedTests: 2 }, + { runnerError: 'spawn failed' }, { executedTests: 0, skippedTests: 0 }]) { + expect(c(over).outcome, JSON.stringify(over)).toBeNull(); + } + }); + + test('slice exit: rule shards stay strict; failed trials never red the runner, missing records do', () => { + const trial = (outcome: 'passed' | 'failed' | null) => ({ status: outcome === 'failed' ? 'failed' as const : 'passed' as const, + trial: { case: 'c', trial: 1, ...plan, outcome, cost_usd: 0, duration_ms: 1, ...(outcome === null ? { harness: 'never started' } : {}) } }); + expect(sliceExitCode([{ status: 'passed' }, trial('failed')])).toBe(0); + expect(sliceExitCode([{ status: 'failed' }, trial('passed')])).toBe(1); + expect(sliceExitCode([{ status: 'passed' }, trial(null)])).toBe(1); + expect(sliceExitCode([{ status: 'timed-out' }])).toBe(1); + const hollow = guardTrialRecords([{ ...trial('passed'), status: 'passed-empty' as const }]); + expect(hollow[0]!.trial!.outcome).toBeNull(); + expect(sliceExitCode(hollow)).toBe(1); + }); + + test('runPaidShards binds each trial to its case, index and panel and records its outcome', async () => { + const evalDirBase = fs.mkdtempSync(path.join(os.tmpdir(), 'trial-shards-')); + try { + const script = (fail: boolean) => `const fs = require('fs'), path = require('path'); +const dir = process.env.GSTACK_EVAL_DIR; fs.mkdirSync(dir, { recursive: true }); +const env = Object.fromEntries(Object.entries(process.env).filter(([k]) => k.startsWith('GSTACK_EVAL_') || k === 'EVALS_SELECTION_JSON')); +fs.writeFileSync(path.join(dir, 'env.json'), JSON.stringify(env)); +fs.writeFileSync(path.join(dir, 'run.json'), JSON.stringify({ tests: [{ name: 'review-sql-injection', passed: ${!fail}, cost_usd: 0.5, + duration_ms: 1, exit_reason: ${fail ? "'timeout'" : "'success'"}, timeout_at_turn: 4, model: 'm' }] })); +console.log("Ran 1 tests across 1 files. [1ms]"); process.exit(${fail ? 1 : 0});`; + const summary = await runPaidShards([[key(1)], [key(2)], [key(3)]], { + jobs: 3, evalDirBase, log: () => {}, trials: { [key(1)]: plan, [key(2)]: plan, [key(3)]: plan }, + commandFor: files => ({ command: process.execPath, args: ['-e', script(files[0] === key(2))] }), + }); + const byKey = (n: number) => summary.outcomes.find(o => o.files[0] === key(n))!; + expect(byKey(1).trial).toMatchObject({ case: 'review-sql-injection', trial: 1, outcome: 'passed', cost_usd: 0.5, model: 'm' }); + expect(byKey(2).trial).toMatchObject({ trial: 2, outcome: 'failed', failure_class: 'timeout', exit_reason: 'timeout', timeout_at_turn: 4 }); + expect(sliceExitCode(summary.outcomes)).toBe(0); + const env = JSON.parse(fs.readFileSync(path.join(evalDirBase, 'shards', shardSlug([key(3)]), 'env.json'), 'utf8')); + expect(env).toMatchObject({ GSTACK_EVAL_CASE_ID: 'review-sql-injection', GSTACK_EVAL_KIND: 'behavior', GSTACK_EVAL_TRIAL: '3', + GSTACK_EVAL_PANEL_N: '3', GSTACK_EVAL_PANEL_K: '2', GSTACK_EVAL_POLICY_VERSION: '1' }); + expect(JSON.parse(env.EVALS_SELECTION_JSON).selected).toEqual(['review-sql-injection']); + } finally { fs.rmSync(evalDirBase, { recursive: true, force: true }); } + }); + + test('file shards exclude isolated names; JUnit cases map to registry ids or stay unattributed', () => { + const pattern = new RegExp(excludedCasesNamePattern(['review-sql-injection'])); + expect(pattern.test('suite > review-sql-injection')).toBe(false); + expect(pattern.test('suite > review-enum-completeness')).toBe(true); + const cases = parseJUnitCases(` + + + +`); + expect(cases).toEqual([ + { name: 'review-sql-injection', classname: 's', outcome: 'passed', timeMs: 1500 }, + { name: 'review-enum-completeness', classname: 's', outcome: 'failed', timeMs: 100, failureType: 'TimeoutError', message: 'test & timed out' }, + { name: 'plain helper', classname: '', outcome: 'skipped', timeMs: 0 }, + ]); + expect(caseIdForTestName('review-sql-injection')).toBe('review-sql-injection'); + expect(caseIdForTestName(CASE_TEST_NAMES['plan-review-report']!)).toBe('plan-review-report'); + expect(caseIdForTestName('plain helper')).toBeNull(); + }); +}); From d5876efa5d73ebbb68ca2cc4e961987766678a0b Mon Sep 17 00:00:00 2001 From: garrytan Date: Tue, 29 Sep 2026 19:30:43 +0000 Subject: [PATCH 11/15] chore(evals): refresh paid duration seeds from proof runs 36597762183 and 36606688266 Both tiers, merged in run order (the later run wins). Notable: split-overflow 1332s -> 504s, section-loading 604s -> 342s, mode-routing 575s -> 444s; multi-finding-batching 734s -> 1318s (its red path in run 36606688266). --- scripts/paid-test-durations.json | 283 ++++++++++++++++--------------- 1 file changed, 147 insertions(+), 136 deletions(-) diff --git a/scripts/paid-test-durations.json b/scripts/paid-test-durations.json index 9ce3dceb3..ca65aceb3 100644 --- a/scripts/paid-test-durations.json +++ b/scripts/paid-test-durations.json @@ -1,190 +1,201 @@ { "version": 2, - "recordedAt": "2026-09-28T09:05:00Z", - "source": "periodic census run 36385945043 (eval-slices = periodic, gate-census = gate); timed-out shards record their wall, self-skipping shards 1s; case shards (#) from the Bun per-case walls in that run (gate census log) and the eval JSON slowest cases (periodic)", + "recordedAt": "2026-09-29T19:29:33.835Z", "tiers": { "gate": { "test/llm-judge-recommendation.test.ts": 1000, - "test/skill-e2e-ask-user-question-format-compliance.test.ts": 60000, + "test/skill-e2e-ask-user-question-format-compliance.test.ts": 63904, "test/skill-e2e-autoplan-dual-voice.test.ts": 1000, - "test/skill-e2e-bws.test.ts": 82000, + "test/skill-e2e-bws.test.ts": 101092, "test/skill-e2e-context-skills.test.ts": 1000, - "test/skill-e2e-coverage-audit.test.ts": 46000, - "test/skill-e2e-cso.test.ts": 251000, - "test/skill-e2e-deploy.test.ts": 427000, + "test/skill-e2e-coverage-audit.test.ts": 50849, + "test/skill-e2e-cso.test.ts": 227394, + "test/skill-e2e-deploy.test.ts": 423978, "test/skill-e2e-design.test.ts": 261000, - "test/skill-e2e-design.test.ts#design-review-detector-shim": 51000, - "test/skill-e2e-design.test.ts#design-review-detector-shim-dom": 108000, - "test/skill-e2e-design.test.ts#design-review-plugin-handoff": 110000, - "test/skill-e2e-design.test.ts#plan-design-review-no-ui-scope": 43000, - "test/skill-e2e-diagram.test.ts": 32000, - "test/skill-e2e-docsync-spawned.test.ts": 51000, + "test/skill-e2e-design.test.ts#design-review-detector-shim": 42811, + "test/skill-e2e-design.test.ts#design-review-detector-shim-dom": 81914, + "test/skill-e2e-design.test.ts#design-review-plugin-handoff": 126642, + "test/skill-e2e-design.test.ts#plan-design-review-no-ui-scope": 24431, + "test/skill-e2e-diagram.test.ts": 29998, + "test/skill-e2e-docsync-spawned.test.ts": 137929, "test/skill-e2e-first-task-scaffold.test.ts": 1000, "test/skill-e2e-gbrain-roundtrip-local.test.ts": 1000, - "test/skill-e2e-hermetic-canary.test.ts": 8000, - "test/skill-e2e-investigate-owned-completion.test.ts": 45000, - "test/skill-e2e-investigate-owned-termination.test.ts": 44000, + "test/skill-e2e-hermetic-canary.test.ts": 8813, + "test/skill-e2e-investigate-owned-completion.test.ts": 45178, + "test/skill-e2e-investigate-owned-termination.test.ts": 48608, "test/skill-e2e-ios-device.test.ts": 1000, - "test/skill-e2e-learnings.test.ts": 35000, - "test/skill-e2e-office-hours-auto-mode.test.ts": 67000, + "test/skill-e2e-learnings.test.ts": 28899, + "test/skill-e2e-office-hours-auto-mode.test.ts": 83101, "test/skill-e2e-office-hours-brain-writeback.test.ts": 1000, "test/skill-e2e-office-hours-phase4.test.ts": 1000, "test/skill-e2e-office-hours.test.ts": 1000, - "test/skill-e2e-plan-ceo-finding-floor.test.ts": 516000, - "test/skill-e2e-plan-ceo-plan-mode.test.ts": 39000, + "test/skill-e2e-plan-ceo-finding-floor.test.ts": 297738, + "test/skill-e2e-plan-ceo-plan-mode.test.ts": 36812, "test/skill-e2e-plan-decision-classification.test.ts": 1000, - "test/skill-e2e-plan-design-with-ui.test.ts": 500000, - "test/skill-e2e-plan-devex-finding-floor.test.ts": 233000, + "test/skill-e2e-plan-design-with-ui.test.ts": 595151, + "test/skill-e2e-plan-devex-finding-floor.test.ts": 207635, "test/skill-e2e-plan-devex-peer-comparison-classification.test.ts": 1000, - "test/skill-e2e-plan-devex-plan-mode.test.ts": 85000, + "test/skill-e2e-plan-devex-plan-mode.test.ts": 115679, "test/skill-e2e-plan-format.test.ts": 1000, - "test/skill-e2e-plan-mode-no-op.test.ts": 206000, + "test/skill-e2e-plan-mode-no-op.test.ts": 217184, "test/skill-e2e-plan-prosons.test.ts": 1000, - "test/skill-e2e-plan-tune.test.ts": 60000, + "test/skill-e2e-plan-tune.test.ts": 63925, "test/skill-e2e-plan.test.ts": 251000, - "test/skill-e2e-plan.test.ts#codex-offered-ceo-review": 55000, - "test/skill-e2e-plan.test.ts#codex-offered-design-review": 59000, - "test/skill-e2e-plan.test.ts#codex-offered-eng-review": 58000, - "test/skill-e2e-plan.test.ts#codex-offered-office-hours": 54000, - "test/skill-e2e-plan.test.ts#office-hours-spec-review": 34000, - "test/skill-e2e-plan.test.ts#plan-ceo-review-benefits": 52000, - "test/skill-e2e-plan.test.ts#plan-review-report": 51000, + "test/skill-e2e-plan.test.ts#codex-offered-ceo-review": 49880, + "test/skill-e2e-plan.test.ts#codex-offered-design-review": 54895, + "test/skill-e2e-plan.test.ts#codex-offered-eng-review": 59818, + "test/skill-e2e-plan.test.ts#codex-offered-office-hours": 50442, + "test/skill-e2e-plan.test.ts#office-hours-spec-review": 33866, + "test/skill-e2e-plan.test.ts#plan-ceo-review-benefits": 48811, + "test/skill-e2e-plan.test.ts#plan-review-report": 76406, "test/skill-e2e-qa-bugs.test.ts": 1000, - "test/skill-e2e-qa-workflow.test.ts": 398000, - "test/skill-e2e-retro.test.ts": 152000, + "test/skill-e2e-qa-callers.test.ts": 737164, + "test/skill-e2e-qa-functional-fix.test.ts": 235133, + "test/skill-e2e-qa-functional.test.ts": 356641, + "test/skill-e2e-qa-workflow.test.ts": 427303, + "test/skill-e2e-retro.test.ts": 141804, "test/skill-e2e-review-army.test.ts": 520000, - "test/skill-e2e-review-army.test.ts#review-army-delivery-audit": 48000, - "test/skill-e2e-review-army.test.ts#review-army-json-findings": 24000, - "test/skill-e2e-review-army.test.ts#review-army-migration-safety": 140000, - "test/skill-e2e-review-army.test.ts#review-army-perf-n-plus-one": 244000, - "test/skill-e2e-review-army.test.ts#review-army-quality-score": 64000, - "test/skill-e2e-review-attribution.test.ts": 87000, - "test/skill-e2e-review.test.ts": 129000, - "test/skill-e2e-session-intelligence.test.ts": 59000, + "test/skill-e2e-review-army.test.ts#review-army-delivery-audit": 50219, + "test/skill-e2e-review-army.test.ts#review-army-json-findings": 21332, + "test/skill-e2e-review-army.test.ts#review-army-migration-safety": 145112, + "test/skill-e2e-review-army.test.ts#review-army-perf-n-plus-one": 241299, + "test/skill-e2e-review-army.test.ts#review-army-quality-score": 86836, + "test/skill-e2e-review-attribution.test.ts": 77592, + "test/skill-e2e-review.test.ts": 136132, + "test/skill-e2e-session-intelligence.test.ts": 55332, "test/skill-e2e-shared-libs-paths.test.ts": 680000, - "test/skill-e2e-shared-libs-paths.test.ts#shared-libs-review-index-flags": 201000, - "test/skill-e2e-shared-libs-paths.test.ts#shared-libs-review-path-eligibility": 252000, - "test/skill-e2e-shared-libs-paths.test.ts#shared-libs-review-prior-coverage": 227000, + "test/skill-e2e-shared-libs-paths.test.ts#shared-libs-review-index-flags": 190202, + "test/skill-e2e-shared-libs-paths.test.ts#shared-libs-review-path-eligibility": 205615, + "test/skill-e2e-shared-libs-paths.test.ts#shared-libs-review-prior-coverage": 214728, "test/skill-e2e-shared-libs.test.ts": 658000, - "test/skill-e2e-shared-libs.test.ts#shared-libs-read-only": 230000, - "test/skill-e2e-shared-libs.test.ts#shared-libs-review-lifecycle": 278000, - "test/skill-e2e-shared-libs.test.ts#shared-libs-review-revalidation": 427000, - "test/skill-e2e-shared-libs.test.ts#shared-libs-unsupported-git": 168000, + "test/skill-e2e-shared-libs.test.ts#shared-libs-read-only": 197454, + "test/skill-e2e-shared-libs.test.ts#shared-libs-review-lifecycle": 267080, + "test/skill-e2e-shared-libs.test.ts#shared-libs-review-revalidation": 316752, + "test/skill-e2e-shared-libs.test.ts#shared-libs-unsupported-git": 170776, "test/skill-e2e-ship-docsync.test.ts": 129000, - "test/skill-e2e-ship-docsync.test.ts#ship-docsync-completion": 491000, - "test/skill-e2e-ship-docsync.test.ts#ship-docsync-current": 364000, - "test/skill-e2e-ship-docsync.test.ts#ship-docsync-failure": 244000, - "test/skill-e2e-ship-docsync.test.ts#ship-docsync-late-result": 169000, - "test/skill-e2e-ship-docsync.test.ts#ship-docsync-launch-failure": 109000, - "test/skill-e2e-ship-docsync.test.ts#ship-docsync-missing-asset": 173000, - "test/skill-e2e-ship-docsync.test.ts#ship-docsync-missing-marker": 131000, - "test/skill-e2e-ship-docsync.test.ts#ship-docsync-recovery": 245000, - "test/skill-e2e-ship-docsync.test.ts#ship-docsync-stale-after": 285000, - "test/skill-e2e-ship-docsync.test.ts#ship-docsync-stale-before": 210000, - "test/skill-e2e-ship-docsync.test.ts#ship-docsync-store": 366000, - "test/skill-e2e-ship-docsync.test.ts#ship-docsync-timeout-unsettled": 151000, - "test/skill-e2e-ship-hook-consent.test.ts": 63000, - "test/skill-e2e-ship-hook-refresh.test.ts": 70000, - "test/skill-e2e-skillify.test.ts": 188000, - "test/skill-e2e-third-party-actions.test.ts": 70000, - "test/skill-e2e-triage.test.ts": 80000, - "test/skill-e2e-workflow.test.ts": 300000, + "test/skill-e2e-ship-docsync.test.ts#ship-docsync-completion": 366116, + "test/skill-e2e-ship-docsync.test.ts#ship-docsync-current": 434135, + "test/skill-e2e-ship-docsync.test.ts#ship-docsync-failure": 164667, + "test/skill-e2e-ship-docsync.test.ts#ship-docsync-late-result": 214218, + "test/skill-e2e-ship-docsync.test.ts#ship-docsync-launch-failure": 140987, + "test/skill-e2e-ship-docsync.test.ts#ship-docsync-missing-asset": 142564, + "test/skill-e2e-ship-docsync.test.ts#ship-docsync-missing-marker": 117648, + "test/skill-e2e-ship-docsync.test.ts#ship-docsync-recovery": 225432, + "test/skill-e2e-ship-docsync.test.ts#ship-docsync-stale-after": 269572, + "test/skill-e2e-ship-docsync.test.ts#ship-docsync-stale-before": 197633, + "test/skill-e2e-ship-docsync.test.ts#ship-docsync-store": 396588, + "test/skill-e2e-ship-docsync.test.ts#ship-docsync-timeout-unsettled": 152278, + "test/skill-e2e-ship-hook-consent.test.ts": 69995, + "test/skill-e2e-ship-hook-refresh.test.ts": 70007, + "test/skill-e2e-ship-skip.test.ts": 80318, + "test/skill-e2e-skillify.test.ts": 191127, + "test/skill-e2e-third-party-actions.test.ts": 82102, + "test/skill-e2e-triage.test.ts": 54088, + "test/skill-e2e-workflow.test.ts": 184925, "test/skill-llm-eval.test.ts": 354000, "test/skill-routing-e2e.test.ts": 1000 }, "periodic": { - "test/carve-section-loading-browse.test.ts": 109000, - "test/carve-section-loading-codex.test.ts": 118000, - "test/carve-section-loading-design-consultation.test.ts": 349000, - "test/carve-section-loading-design-html.test.ts": 259000, - "test/carve-section-loading-design-shotgun.test.ts": 194000, - "test/carve-section-loading-document-release.test.ts": 95000, - "test/carve-section-loading-land-and-deploy.test.ts": 185000, - "test/carve-section-loading-plan-design-review.test.ts": 284000, - "test/carve-section-loading-plan-devex-review.test.ts": 396000, - "test/carve-section-loading-plan-eng-review.test.ts": 301000, - "test/carve-section-loading-qa.test.ts": 114000, - "test/carve-section-loading-retro.test.ts": 230000, - "test/carve-section-loading-review.test.ts": 174000, - "test/carve-section-loading-setup-gbrain.test.ts": 91000, - "test/carve-section-loading-spec.test.ts": 150000, + "test/carve-section-loading-browse.test.ts": 92911, + "test/carve-section-loading-codex.test.ts": 150172, + "test/carve-section-loading-design-consultation.test.ts": 318385, + "test/carve-section-loading-design-html.test.ts": 245660, + "test/carve-section-loading-design-shotgun.test.ts": 140298, + "test/carve-section-loading-document-release.test.ts": 97810, + "test/carve-section-loading-land-and-deploy.test.ts": 156293, + "test/carve-section-loading-plan-design-review.test.ts": 223606, + "test/carve-section-loading-plan-devex-review.test.ts": 384295, + "test/carve-section-loading-plan-eng-review.test.ts": 297627, + "test/carve-section-loading-qa.test.ts": 149637, + "test/carve-section-loading-retro.test.ts": 199931, + "test/carve-section-loading-review.test.ts": 247902, + "test/carve-section-loading-setup-gbrain.test.ts": 80994, + "test/carve-section-loading-spec.test.ts": 176639, "test/codex-e2e-recommendation-substance.test.ts": 1000, "test/codex-e2e-shared-libs.test.ts": 1000, "test/codex-e2e-sol-scope.test.ts": 1000, "test/codex-e2e.test.ts": 1000, - "test/llm-judge-recommendation.test.ts": 15000, - "test/skill-e2e-arm-benchmark.test.ts": 76000, + "test/llm-judge-recommendation.test.ts": 15334, + "test/skill-e2e-arm-benchmark.test.ts": 75588, "test/skill-e2e-aside.test.ts": 1000, - "test/skill-e2e-auq-consistency.test.ts": 75000, - "test/skill-e2e-auq-matrix.test.ts": 320000, - "test/skill-e2e-auq-verbose-vs-carved-ab.test.ts": 66000, - "test/skill-e2e-auto-decide-preserved.test.ts": 113000, - "test/skill-e2e-autoplan-dual-voice.test.ts": 532000, - "test/skill-e2e-benchmark-providers.test.ts": 9000, + "test/skill-e2e-auq-consistency.test.ts": 70429, + "test/skill-e2e-auq-matrix.test.ts": 165796, + "test/skill-e2e-auq-verbose-vs-carved-ab.test.ts": 58547, + "test/skill-e2e-auto-decide-preserved.test.ts": 154931, + "test/skill-e2e-autoplan-dual-voice.test.ts": 167918, + "test/skill-e2e-benchmark-providers.test.ts": 9495, "test/skill-e2e-bws.test.ts": 1000, - "test/skill-e2e-context-skills.test.ts": 181000, + "test/skill-e2e-context-skills.test.ts": 169497, "test/skill-e2e-coverage-audit.test.ts": 1000, - "test/skill-e2e-cso.test.ts": 358000, + "test/skill-e2e-cso.test.ts": 253498, "test/skill-e2e-deploy.test.ts": 1000, "test/skill-e2e-design.test.ts": 817000, - "test/skill-e2e-design.test.ts#design-consultation-core": 204000, - "test/skill-e2e-design.test.ts#design-html-slop-gate": 205000, - "test/skill-e2e-design.test.ts#plan-design-review-plan-mode": 283000, - "test/skill-e2e-diagram.test.ts": 166000, - "test/skill-e2e-first-task-scaffold.test.ts": 9000, + "test/skill-e2e-design.test.ts#design-consultation-core": 159257, + "test/skill-e2e-design.test.ts#design-consultation-existing": 175607, + "test/skill-e2e-design.test.ts#design-consultation-preview": 114638, + "test/skill-e2e-design.test.ts#design-consultation-research": 109573, + "test/skill-e2e-design.test.ts#design-html-slop-gate": 158970, + "test/skill-e2e-design.test.ts#plan-design-review-plan-mode": 300166, + "test/skill-e2e-diagram.test.ts": 55187, + "test/skill-e2e-first-task-scaffold.test.ts": 8563, "test/skill-e2e-gbrain-roundtrip-local.test.ts": 1000, - "test/skill-e2e-health.test.ts": 165000, + "test/skill-e2e-health.test.ts": 188584, "test/skill-e2e-hermetic-canary.test.ts": 1000, "test/skill-e2e-ios-device.test.ts": 1000, "test/skill-e2e-learnings.test.ts": 1000, - "test/skill-e2e-office-hours-brain-writeback.test.ts": 174000, - "test/skill-e2e-office-hours-phase4.test.ts": 60000, + "test/skill-e2e-office-hours-brain-writeback.test.ts": 203044, + "test/skill-e2e-office-hours-design-draft.test.ts": 285868, + "test/skill-e2e-office-hours-phase4.test.ts": 41421, "test/skill-e2e-office-hours-section-loading.test.ts": 1200000, - "test/skill-e2e-office-hours.test.ts": 119000, - "test/skill-e2e-outside-plan-disabled.test.ts": 43000, + "test/skill-e2e-office-hours.test.ts": 136970, + "test/skill-e2e-outside-plan-disabled.test.ts": 53292, "test/skill-e2e-outside-voice.test.ts": 1000, - "test/skill-e2e-overlay-harness-claude-dedicated-tools-vs-bash-sonnet.test.ts": 81000, - "test/skill-e2e-overlay-harness-claude-dedicated-tools-vs-bash.test.ts": 66000, - "test/skill-e2e-overlay-harness-opus-4-7-effort-match-trivial.test.ts": 42000, - "test/skill-e2e-overlay-harness-opus-4-7-literal-interpretation.test.ts": 278000, - "test/skill-e2e-plan-ceo-mode-routing.test.ts": 575000, - "test/skill-e2e-plan-ceo-review-section-loading.test.ts": 604000, - "test/skill-e2e-plan-ceo-split-overflow.test.ts": 1332000, - "test/skill-e2e-plan-decision-classification.test.ts": 122000, - "test/skill-e2e-plan-design-finding-floor.test.ts": 175000, - "test/skill-e2e-plan-design-plan-mode.test.ts": 117000, - "test/skill-e2e-plan-devex-peer-comparison-classification.test.ts": 83000, - "test/skill-e2e-plan-eng-finding-floor.test.ts": 282000, - "test/skill-e2e-plan-eng-multi-finding-batching.test.ts": 734000, - "test/skill-e2e-plan-eng-plan-mode.test.ts": 91000, - "test/skill-e2e-plan-format.test.ts": 282000, - "test/skill-e2e-plan-prosons.test.ts": 180000, + "test/skill-e2e-overlay-harness-claude-dedicated-tools-vs-bash-sonnet.test.ts": 85282, + "test/skill-e2e-overlay-harness-claude-dedicated-tools-vs-bash.test.ts": 80465, + "test/skill-e2e-overlay-harness-opus-4-7-effort-match-trivial.test.ts": 49282, + "test/skill-e2e-overlay-harness-opus-4-7-literal-interpretation.test.ts": 318532, + "test/skill-e2e-plan-ceo-mode-routing.test.ts": 443504, + "test/skill-e2e-plan-ceo-review-section-loading.test.ts": 342250, + "test/skill-e2e-plan-ceo-split-overflow.test.ts": 504266, + "test/skill-e2e-plan-decision-classification.test.ts": 93469, + "test/skill-e2e-plan-design-finding-floor.test.ts": 155538, + "test/skill-e2e-plan-design-plan-mode.test.ts": 106608, + "test/skill-e2e-plan-devex-peer-comparison-classification.test.ts": 79786, + "test/skill-e2e-plan-eng-finding-floor.test.ts": 388666, + "test/skill-e2e-plan-eng-multi-finding-batching.test.ts": 1318488, + "test/skill-e2e-plan-eng-plan-mode.test.ts": 229226, + "test/skill-e2e-plan-format.test.ts": 219670, + "test/skill-e2e-plan-prosons.test.ts": 163659, "test/skill-e2e-plan-tune.test.ts": 1000, "test/skill-e2e-plan.test.ts": 824000, - "test/skill-e2e-plan.test.ts#plan-ceo-review": 123000, - "test/skill-e2e-plan.test.ts#plan-ceo-review-selective": 244000, - "test/skill-e2e-plan.test.ts#plan-eng-review-artifact": 153000, - "test/skill-e2e-qa-bugs.test.ts": 283000, - "test/skill-e2e-qa-workflow.test.ts": 239000, - "test/skill-e2e-retro.test.ts": 114000, + "test/skill-e2e-plan.test.ts#plan-ceo-review": 136101, + "test/skill-e2e-plan.test.ts#plan-ceo-review-expansion-energy": 91984, + "test/skill-e2e-plan.test.ts#plan-ceo-review-selective": 267212, + "test/skill-e2e-plan.test.ts#plan-eng-review": 128506, + "test/skill-e2e-plan.test.ts#plan-eng-review-artifact": 164137, + "test/skill-e2e-qa-bugs.test.ts": 333847, + "test/skill-e2e-qa-workflow.test.ts": 403953, + "test/skill-e2e-retro.test.ts": 166711, "test/skill-e2e-review-army.test.ts": 511000, - "test/skill-e2e-review-army.test.ts#review-army-consensus": 278000, - "test/skill-e2e-review-army.test.ts#review-army-simplification": 144000, + "test/skill-e2e-review-army.test.ts#review-army-consensus": 277908, + "test/skill-e2e-review-army.test.ts#review-army-red-team": 88882, + "test/skill-e2e-review-army.test.ts#review-army-simplification": 136644, + "test/skill-e2e-review-army.test.ts#review-army-simplification-precision": 23884, "test/skill-e2e-review-attribution.test.ts": 1000, - "test/skill-e2e-review.test.ts": 131000, + "test/skill-e2e-review.test.ts": 179362, "test/skill-e2e-session-intelligence.test.ts": 1000, - "test/skill-e2e-setup-gbrain-bad-token.test.ts": 49000, - "test/skill-e2e-setup-gbrain-path4-local-pglite.test.ts": 119000, - "test/skill-e2e-setup-gbrain-remote.test.ts": 110000, - "test/skill-e2e-shared-libs-periodic.test.ts": 415000, - "test/skill-e2e-ship-section-loading.test.ts": 379000, - "test/skill-e2e-skillify.test.ts": 56000, - "test/skill-e2e-sync-gbrain-readiness.test.ts": 74000, + "test/skill-e2e-setup-gbrain-bad-token.test.ts": 42912, + "test/skill-e2e-setup-gbrain-path4-local-pglite.test.ts": 165743, + "test/skill-e2e-setup-gbrain-remote.test.ts": 121127, + "test/skill-e2e-shared-libs-periodic.test.ts": 369660, + "test/skill-e2e-ship-section-loading.test.ts": 255273, + "test/skill-e2e-skillify.test.ts": 58522, + "test/skill-e2e-sync-gbrain-readiness.test.ts": 60234, "test/skill-e2e-third-party-actions.test.ts": 1000, "test/skill-e2e-triage.test.ts": 1000, "test/skill-e2e-workflow.test.ts": 1000, - "test/skill-llm-eval.test.ts": 402000, - "test/skill-routing-e2e.test.ts": 57000 + "test/skill-llm-eval.test.ts": 383454, + "test/skill-routing-e2e.test.ts": 68843 } } } From 62fb9a255d3c9737cbe390d6b5e2ab0e4a4fffd3 Mon Sep 17 00:00:00 2001 From: garrytan Date: Tue, 29 Sep 2026 19:30:43 +0000 Subject: [PATCH 12/15] feat(evals): stamp trial series identities and fit panels to the live registry - scripts/eval-trial-series.ts stamps series_identity (eval-flake-rank's caseSeriesIdentities) on a report's trial-outcomes JSONL as its own step, keeping the history tool out of the paid runner's closure; TrialOutcomeRecord gains the optional series_identity field. - Slice-count plans let a registered trial spill into an ordinary lane when its siblings hold every long lane, so panels never share a runner. - Re-audited test-selection.ts (Stream B added the E2E_KINDS/BEHAVIOR_WHY map-diff; no new module loading) and repinned its hash. - Detach and release floors now count trial shards (66 periodic trials in 22 panels): periodic floor 33,821s, still under eval:bg:periodic's 67,380s. - Coordination fixtures supply the executor's trial records. --- scripts/eval-trial-series.ts | 35 +++++++++++++++++++++++++++ scripts/test-paid-shards.ts | 34 +++++++++++--------------- test/ci-paid-coordination.test.ts | 10 ++++++-- test/eng-finding-retry-budget.test.ts | 10 +++++--- test/helpers/eval-store.ts | 3 +++ test/paid-free-boundary.test.ts | 2 +- test/paid-report-fail-open.test.ts | 9 +++++-- test/paid-retry-supervision.test.ts | 4 +-- 8 files changed, 76 insertions(+), 31 deletions(-) create mode 100644 scripts/eval-trial-series.ts diff --git a/scripts/eval-trial-series.ts b/scripts/eval-trial-series.ts new file mode 100644 index 000000000..e6b370f83 --- /dev/null +++ b/scripts/eval-trial-series.ts @@ -0,0 +1,35 @@ +#!/usr/bin/env bun +/** + * Stamp `series_identity` on a report's trial-outcomes JSONL (the pass-rates + * history key: a hash of each case's own touchfiles, GLOBAL_TOUCHFILES + * excluded; scripts/eval-flake-rank.ts caseSeriesIdentities). A separate step + * after `test-paid-shards.ts --report`, so the paid runner's closure never + * imports the history tool. + * + * Usage: bun run scripts/eval-trial-series.ts + */ +import * as fs from 'node:fs'; +import * as path from 'node:path'; +import { caseSeriesIdentities } from './eval-flake-rank'; +import { formatTrialOutcomes, parseTrialOutcomes } from '../test/helpers/eval-store'; + +const ROOT = path.resolve(import.meta.dir, '..'); + +/** Rewrite the file with every record stamped; an invalid line fails the whole stamp. */ +export function stampTrialSeries(file: string, root = ROOT): number { + const { records, errors } = parseTrialOutcomes(fs.readFileSync(file, 'utf8')); + if (errors.length) throw new Error(`${file}: ${errors.join('; ')}`); + const identities = caseSeriesIdentities([...new Set(records.map(record => record.case))], root); + const stamped = records.map(record => ({ ...record, series_identity: identities[record.case] })); + fs.writeFileSync(file, formatTrialOutcomes(stamped)); + return stamped.length; +} + +if (import.meta.main) { + const file = process.argv[2]; + if (!file) { + console.error('usage: bun run scripts/eval-trial-series.ts '); + process.exit(2); + } + console.log(`[eval-trial-series] stamped ${stampTrialSeries(file)} record(s) in ${file}`); +} diff --git a/scripts/test-paid-shards.ts b/scripts/test-paid-shards.ts index 4fdd84b8e..198f8aceb 100644 --- a/scripts/test-paid-shards.ts +++ b/scripts/test-paid-shards.ts @@ -1739,11 +1739,18 @@ export function buildRunManifest(opts: { ordinary.filter(files => !registeredFiles.has(files[0])))) { const lanes = registeredFiles.has(files[0]) ? longLanes : ordinarySlices; const laneKeys = (index: number) => [...allocations].filter(([, lane]) => lane === index + 1).map(([key]) => key); - let lane = -1; - for (let index = 0; index < lanes; index++) { - if (sharesPanel(laneKeys(index), files[0]!)) continue; - if (lane < 0 || loads[index] < loads[lane]) lane = index; - } + // A trial whose siblings already hold every long lane may use any + // ordinary lane: independent runners outrank long-lane ownership. + const pick = (limit: number) => { + let best = -1; + for (let index = 0; index < limit; index++) { + if (sharesPanel(laneKeys(index), files[0]!)) continue; + if (best < 0 || loads[index] < loads[best]) best = index; + } + return best; + }; + let lane = pick(lanes); + if (lane < 0) lane = pick(ordinarySlices); if (lane < 0) lane = loads.slice(0, lanes).indexOf(Math.min(...loads.slice(0, lanes))); allocations.set(files[0], lane + 1); loads[lane] += resolvePaidShardTimeoutMs(files, opts.timeoutMs); @@ -2500,21 +2507,7 @@ export function runPaidReport(reportDir: string, options: { writeDurations?: boo const laterPanels = attempts.slice(1).flatMap(attempt => panelReports(manifest, artifacts.map(a => a.result), attempt) .filter(panel => panel.trials.length > 0)); - // Quarantine policy checks on census runs: the per-tier cap and entry expiry. - if (manifest.evalsAll) { - const tierIds = Object.keys(E2E_TIERS).filter(id => E2E_TIERS[id] === manifest.tier); - const quarantined = Object.keys(CASE_QUARANTINE).filter(id => E2E_TIERS[id] === manifest.tier); - if (quarantined.length > EVAL_POLICY.quarantine.capFraction * tierIds.length) { - verdict.problems.push(`QUARANTINE over cap: ${quarantined.length} of ${tierIds.length} ${manifest.tier} cases (cap ${Math.round(EVAL_POLICY.quarantine.capFraction * 100)}%)`); - } - const expiryMs = EVAL_POLICY.quarantine.expiryWeeklyRuns * 7 * 24 * 60 * 60 * 1000; - for (const id of quarantined) { - const entered = Date.parse(CASE_QUARANTINE[id]!.enteredAt); - if (!Number.isFinite(entered) || Date.now() - entered > expiryMs) { - verdict.problems.push(`QUARANTINE expired: ${id} (entered ${CASE_QUARANTINE[id]!.enteredAt}; entries expire after ${EVAL_POLICY.quarantine.expiryWeeklyRuns} weekly runs)`); - } - } - } + // Quarantine cap and expiry are the weekly pass-rates gate's (eval-flake-rank --gate). // History: one trial-outcomes line per isolated trial and per JUnit rule/judge case. const runId = env.GITHUB_RUN_ID; @@ -2567,6 +2560,7 @@ export function runPaidReport(reportDir: string, options: { writeDurations?: boo } } } + // series_identity is stamped afterwards by scripts/eval-trial-series.ts (the report job's next step). fs.writeFileSync(trialOutcomesPath, formatTrialOutcomes(history)); // Headline and failure block (A4): one formatter for the log, the PR comment and the weekly issue. diff --git a/test/ci-paid-coordination.test.ts b/test/ci-paid-coordination.test.ts index 7405037e9..5601e9087 100644 --- a/test/ci-paid-coordination.test.ts +++ b/test/ci-paid-coordination.test.ts @@ -3,7 +3,7 @@ import * as fs from 'node:fs'; import * as os from 'node:os'; import * as path from 'node:path'; import { spawnSync } from 'node:child_process'; -import { buildRunManifest, collectPaidTestFiles, type PaidRunManifest, type SliceResult } from '../scripts/test-paid-shards'; +import { buildRunManifest, collectPaidTestFiles, shardCaseId, shardTrial, type PaidRunManifest, type SliceResult } from '../scripts/test-paid-shards'; import { STRICT_RETRY_CASE_BUDGETS } from './helpers/eval-budgets'; import { approvedCookieWorkflowSource, manualReviewFixture } from './helpers/manual-judge-review-fixture'; @@ -16,6 +16,11 @@ type Job = { permissions: Record; steps: Step[]; }; +/** A passing trial record for an isolated trial shard (the executor's current result schema). */ +const trialRecord = (entry: PaidRunManifest['entries'][number]) => entry.trial ? { trial: { + case: shardCaseId(entry.file)!, trial: shardTrial(entry.file)!, ...entry.trial, outcome: 'passed' as const, cost_usd: 0, duration_ms: 1, +} } : {}; + const workflows = ['evals.yml', 'evals-periodic.yml'].map(name => ({ name, jobs: (Bun.YAML.parse(fs.readFileSync(path.join(ROOT, '.github/workflows', name), 'utf8')) as { @@ -217,6 +222,7 @@ describe('dependency-free CI planner and report execution', () => { executedTests: STRICT_RETRY_CASE_BUDGETS.find(budget => budget.file === entry.file)?.cases ?? 1, skippedTests: 0, ...(entry.budget ? { budget: entry.budget } : {}), + ...trialRecord(entry), })), }; fs.writeFileSync(path.join(reportDir, `slice-${sliceIndex}.json`), JSON.stringify(result)); @@ -275,7 +281,7 @@ describe('dependency-free CI planner and report execution', () => { outcomes: manifest.entries.filter(entry => entry.status === 'planned').map(entry => ({ files: [entry.file], status: 'passed', exitCode: 0, elapsedMs: 1, executedTests: STRICT_RETRY_CASE_BUDGETS.find(budget => budget.file === entry.file)?.cases ?? 1, - skippedTests: 0, ...(entry.budget ? { budget: entry.budget } : {}), + skippedTests: 0, ...(entry.budget ? { budget: entry.budget } : {}), ...trialRecord(entry), })), }; const slicePath = path.join(reportDir, 'slice-1.json'); diff --git a/test/eng-finding-retry-budget.test.ts b/test/eng-finding-retry-budget.test.ts index a865fd8a3..b8aa70b78 100644 --- a/test/eng-finding-retry-budget.test.ts +++ b/test/eng-finding-retry-budget.test.ts @@ -1,5 +1,5 @@ import { expect, test } from 'bun:test'; -import { resolvePaidShardBudget, retriesForFiles, planPaidShards, parseRunManifest, verifySliceResults, runPaidShard, buildRunManifest, paidShardWallUpperBoundMs, collectPaidTestFiles, selectPaidTestFiles, isOverlayTestFile, DEFAULT_SHARD_TIMEOUT_MS, DEFAULT_JOBS, parseCliOptions, expandCaseShards, shardFile, sliceExecutionOrder, sliceSupervisedWallMs } from '../scripts/test-paid-shards'; +import { resolvePaidShardBudget, retriesForFiles, planPaidShards, parseRunManifest, verifySliceResults, runPaidShard, buildRunManifest, paidShardWallUpperBoundMs, collectPaidTestFiles, selectPaidTestFiles, isOverlayTestFile, DEFAULT_SHARD_TIMEOUT_MS, DEFAULT_JOBS, parseCliOptions, expandCaseShards, expandTrialShards, shardFile, sliceExecutionOrder, sliceSupervisedWallMs } from '../scripts/test-paid-shards'; import { FINDING_RETRY_BUDGETS, ALL_TIERS, SHARD_RESERVE_MS } from './helpers/eval-budgets'; import fs from 'node:fs'; import os from 'node:os'; @@ -175,8 +175,9 @@ test('single-slice manifest retains all registered files with one allocation', ( test('current detach supervision covers the live-census floor', () => { const floorFor = (tier: 'gate' | 'periodic') => { - // Case-sharded files contribute one shard per case, exactly as the runner plans. - const files = expandCaseShards(selectPaidTestFiles(collectPaidTestFiles(), tier).selected, tier); + // Case-sharded files contribute one shard per case and isolated cases one + // shard per trial, exactly as the runner plans. + const files = expandTrialShards(expandCaseShards(selectPaidTestFiles(collectPaidTestFiles(), tier).selected, tier), tier).keys; const excess = files.reduce((n, file) => n + Math.max(0, resolvePaidShardBudget([file]).timeoutMs - DEFAULT_SHARD_TIMEOUT_MS), 0); return Math.ceil((Math.ceil(files.length / DEFAULT_JOBS) * DEFAULT_SHARD_TIMEOUT_MS + excess) / 1000 * 1.05); }; @@ -186,7 +187,8 @@ test('current detach supervision covers the live-census floor', () => { expect(floorFor('gate')).toBe(21_725); expect(gateTimeout).toBe(49_320); expect(gateTimeout).toBeGreaterThanOrEqual(floorFor('gate')); - expect(floorFor('periodic')).toBe(22_481); + expect(floorFor('periodic')).toBe(33_821); + expect(periodicTimeout).toBeGreaterThanOrEqual(floorFor('periodic')); }); for (const jobs of [1, 2, 3]) test(`FIFO bound covers partial durations with ${jobs} workers`, () => { diff --git a/test/helpers/eval-store.ts b/test/helpers/eval-store.ts index 32de29324..23ec90ffd 100644 --- a/test/helpers/eval-store.ts +++ b/test/helpers/eval-store.ts @@ -402,6 +402,8 @@ export interface TrialOutcomeRecord { sha?: string; lane?: string; recorded_at?: string; + /** History series key: a hash of the case's own touchfiles (GLOBAL_TOUCHFILES excluded), stamped by the report job. */ + series_identity?: string; } /** First line of free text, stripped of @-mentions and control characters, capped. */ @@ -434,6 +436,7 @@ function trialRecordProblems(r: any): string[] { if (r.execution !== 'executed' && r.execution !== 'reused') problems.push('execution invalid'); if (!['shard', 'junit', 'backfill'].includes(r.source)) problems.push('source invalid'); if (r.error !== undefined && (typeof r.error !== 'string' || r.error.length > TRIAL_ERROR_MAX)) problems.push('error invalid'); + if (r.series_identity !== undefined && (typeof r.series_identity !== 'string' || !/^[\w.-]{1,64}$/.test(r.series_identity))) problems.push('series_identity invalid'); return problems; } diff --git a/test/paid-free-boundary.test.ts b/test/paid-free-boundary.test.ts index 71a5350d3..45d9142ff 100644 --- a/test/paid-free-boundary.test.ts +++ b/test/paid-free-boundary.test.ts @@ -27,7 +27,7 @@ function runnerDependencies(root: string, entries: string[]): string[] { const source = fs.readFileSync(file, 'utf8').replace(/^#![^\n]*(?:\n|$)/, '\n'); let audited = source; if (relative === 'test/helpers/test-selection.ts') { - if (createHash('sha256').update(source).digest('hex') !== '4d2fbcb6249e8d22453d25bfe9b18ee0f4568bbec071918675c38a455d4e1e08') { + if (createHash('sha256').update(source).digest('hex') !== '052ad5a52472bcb41db04c9f21fe6a819e9547768468f7e5d390e0014b567677') { throw new Error('Re-audit the historical touchfile map loader before excluding its computed import'); } audited = source.replace('`const m = await import(${JSON.stringify(dataPath)});`,', "'',"); diff --git a/test/paid-report-fail-open.test.ts b/test/paid-report-fail-open.test.ts index a286c737c..27343310c 100644 --- a/test/paid-report-fail-open.test.ts +++ b/test/paid-report-fail-open.test.ts @@ -11,6 +11,7 @@ import * as os from 'node:os'; import * as path from 'node:path'; import { spawnSync } from 'node:child_process'; import { parseRunManifest, type PaidRunManifest, type SliceResult } from '../scripts/test-paid-shards'; +import { stampTrialSeries } from '../scripts/eval-trial-series'; const ROOT = path.resolve(import.meta.dir, '..'); const RULE_A = 'test/skill-e2e-fail-open-alpha.test.ts'; @@ -148,8 +149,12 @@ describe('behavior and quarantined panels through --report', () => { const summary = JSON.parse(fs.readFileSync(path.join(r.dir, 'collector-outcomes.json'), 'utf8')); expect(summary.version).toBe(2); expect(summary.panels[0]).toMatchObject({ case: ID, status: 'PASS', split: true, failsLane: false }); - const history = fs.readFileSync(path.join(r.dir, 'trial-outcomes.jsonl'), 'utf8').trim().split('\n').map(line => JSON.parse(line)); - expect(history.map(h => [h.trial, h.outcome])).toEqual([[1, 'passed'], [2, 'failed'], [3, 'passed']]); + const outcomesFile = path.join(r.dir, 'trial-outcomes.jsonl'); + const history = () => fs.readFileSync(outcomesFile, 'utf8').trim().split('\n').map(line => JSON.parse(line)); + expect(history().map(h => [h.trial, h.outcome])).toEqual([[1, 'passed'], [2, 'failed'], [3, 'passed']]); + expect(stampTrialSeries(outcomesFile)).toBe(3); + expect(new Set(history().map(h => h.series_identity)).size).toBe(1); + expect(history()[0].series_identity).toMatch(/^[0-9a-f]{16}$/); }); test('behavior 1/3: red', () => { diff --git a/test/paid-retry-supervision.test.ts b/test/paid-retry-supervision.test.ts index cb989a9c2..7f0e93796 100644 --- a/test/paid-retry-supervision.test.ts +++ b/test/paid-retry-supervision.test.ts @@ -230,8 +230,8 @@ test('detached PR fallback and release commands cover their actual default worke )) / 1000 * 1.05)); } const detachedReleaseWall = Number(scripts['eval:bg:release'].match(/--timeout (\d+)/)?.[1]) * 1000; - expect(releaseFloors).toEqual([21_725, 22_481]); - expect(releaseFloors.reduce((total, floor) => total + floor, 0)).toBe(44_206); + expect(releaseFloors).toEqual([21_725, 33_821]); + expect(releaseFloors.reduce((total, floor) => total + floor, 0)).toBe(55_546); expect(detachedReleaseWall).toBe(116_700_000); expect(detachedReleaseWall).toBeGreaterThanOrEqual(releaseWall + 120_000); }); From 8622535b90c462d5cf9c3267c12284eb0b0eb300 Mon Sep 17 00:00:00 2001 From: garrytan Date: Tue, 29 Sep 2026 19:34:33 +0000 Subject: [PATCH 13/15] ci(evals): attempt-scoped artifacts, verdict-v2 PR comment, weekly pass-rate gate and one INFRA re-dispatch - Slice, census and marathon artifacts carry -a; reports download them per artifact (no merge), so records never overwrite and a re-run never replaces the first attempt's verdict. - Planners pass --max-parallel for the capacity preflight (24/16 unchanged: the refreshed periodic plan needs 24 slices, the gate census 12). - PR comment: jq-only job reads collector-outcomes v2 (headline, sanitized failure block); the group_by(.name)|last recomputation is gone. - Reports stamp series identities, upload trial-outcomes-* for history, and shard logs upload always (a failed trial no longer reds its runner). - Weekly report: headline + failure block of both lanes in the issue body, the eval:pass-rates --gate step (fails closed without history), close the issue on a green run, and UC-E1: when every red is machine-classified INFRA/INCOMPLETE, one re-dispatch as a new run in its own concurrency group (redispatch_of), both runs reported. --- .github/workflows/evals-marathon.yml | 7 +- .github/workflows/evals-periodic.yml | 180 ++++++++++++++++++++++----- .github/workflows/evals.yml | 136 +++++++++----------- test/ci-paid-coordination.test.ts | 7 +- test/evals-workflow-wiring.test.ts | 65 ++++++++++ 5 files changed, 284 insertions(+), 111 deletions(-) diff --git a/.github/workflows/evals-marathon.yml b/.github/workflows/evals-marathon.yml index b73a7ea1c..7d3e52326 100644 --- a/.github/workflows/evals-marathon.yml +++ b/.github/workflows/evals-marathon.yml @@ -187,7 +187,7 @@ jobs: if: always() uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 with: - name: marathon-slice-${{ matrix.slice }} + name: marathon-slice-${{ matrix.slice }}-a${{ github.run_attempt }} path: /tmp/marathon-slice-results retention-days: 90 @@ -209,7 +209,7 @@ jobs: if: failure() uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 with: - name: marathon-slice-${{ matrix.slice }}-logs + name: marathon-logs-slice-${{ matrix.slice }}-a${{ github.run_attempt }} include-hidden-files: true # The Fix-bun-temp step points TMPDIR at /home/runner/.cache, so the # runner's spool lands THERE, not /tmp — the original /tmp glob @@ -247,9 +247,8 @@ jobs: - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8 with: - pattern: marathon-slice-[0-9]* + pattern: marathon-slice-* path: /tmp/marathon-report - merge-multiple: true - name: Reconcile slices against the manifest (fail-closed) id: reconcile diff --git a/.github/workflows/evals-periodic.yml b/.github/workflows/evals-periodic.yml index a684b7b5b..707c2221c 100644 --- a/.github/workflows/evals-periodic.yml +++ b/.github/workflows/evals-periodic.yml @@ -19,9 +19,16 @@ on: schedule: - cron: '0 6 * * 1' # Monday 6 AM UTC (ci-image prebuilds at 4 AM) workflow_dispatch: + inputs: + redispatch_of: + description: 'Run id this run re-dispatches (the one INFRA/INCOMPLETE-only re-dispatch; set by the report job)' + type: string + default: '' +# A re-dispatch runs in its own group so it never cancels the run that +# dispatched it; both runs are reported. concurrency: - group: evals-periodic + group: evals-periodic${{ inputs.redispatch_of && format('-redispatch-{0}', inputs.redispatch_of) || '' }} cancel-in-progress: true env: @@ -106,7 +113,7 @@ jobs: - name: Emit run manifest (ALL periodic tests minus reasoned excludes) env: EVALS_ALL: "1" - run: EVALS_TIER=periodic bun --no-install run scripts/test-paid-shards.ts --tier periodic --emit-plan /tmp/paid-plan/manifest.json --slice-budget 540 --jobs 2 + run: EVALS_TIER=periodic bun --no-install run scripts/test-paid-shards.ts --tier periodic --emit-plan /tmp/paid-plan/manifest.json --slice-budget 540 --jobs 2 --max-parallel 24 - name: Derive the periodic executor matrix from the plan id: periodic-matrix @@ -123,7 +130,7 @@ jobs: - name: Emit gate census manifest (ALL gate tests) env: EVALS_ALL: "1" - run: EVALS_TIER=gate bun run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/gate-census-plan/manifest.json --slice-budget 540 --jobs 2 --skip-judges + run: EVALS_TIER=gate bun run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/gate-census-plan/manifest.json --slice-budget 540 --jobs 2 --skip-judges --max-parallel 16 - name: Derive the gate census executor matrix from the plan id: gate-matrix @@ -214,7 +221,7 @@ jobs: if: always() uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 with: - name: paid-slice-${{ matrix.slice }} + name: paid-slice-${{ matrix.slice }}-a${{ github.run_attempt }} path: /tmp/paid-slice-results retention-days: 90 @@ -232,11 +239,13 @@ jobs: if-no-files-found: ignore retention-days: 90 - - name: Upload shard logs on failure - if: failure() + # always(), not failure(): a failed behavior trial is a verdict and no + # longer reds its runner, but its full log is the diagnosis evidence. + - name: Upload shard logs + if: always() uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 with: - name: paid-slice-${{ matrix.slice }}-logs + name: paid-logs-slice-${{ matrix.slice }}-a${{ github.run_attempt }} include-hidden-files: true # The Fix-bun-temp step points TMPDIR at /home/runner/.cache, so the # runner's spool lands THERE, not /tmp — the original /tmp glob @@ -312,7 +321,7 @@ jobs: if: always() uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 with: - name: gate-census-${{ matrix.slice }} + name: gate-census-${{ matrix.slice }}-a${{ github.run_attempt }} path: /tmp/gate-census-results retention-days: 90 @@ -337,12 +346,16 @@ jobs: # missing slice artifact reading as green is the class this lane kills — # but a cancelled run stops here. if: ${{ !cancelled() && needs.plan-slices.result == 'success' }} - timeout-minutes: 10 + timeout-minutes: 15 permissions: contents: read - # The failure notification below upserts a tracking issue via - # `gh api /issues` — gated by the issues permission. + # The notification below upserts (or closes) a tracking issue via + # `gh issue` — gated by the issues permission. issues: write + # Pass-rate history downloads earlier weekly runs' trial-outcomes. + actions: read + outputs: + redispatch: ${{ steps.verdict.outputs.redispatch }} steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7 with: @@ -357,11 +370,12 @@ jobs: name: paid-plan path: /tmp/paid-report + # One directory per attempt-scoped slice artifact (no merge): shard + # records never overwrite each other and the first attempt decides. - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8 with: - pattern: paid-slice-[0-9]* + pattern: paid-slice-* path: /tmp/paid-report - merge-multiple: true - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8 with: @@ -372,7 +386,6 @@ jobs: with: pattern: gate-census-[0-9]* path: /tmp/gate-census-report - merge-multiple: true - name: Reconcile slices against the manifest (fail-closed) id: reconcile @@ -394,34 +407,114 @@ jobs: EVALS_TIER=gate bun run scripts/test-paid-shards.ts --tier gate --report /tmp/gate-census-report | tee /tmp/gate-report.txt echo "exit=${PIPESTATUS[0]}" >> "$GITHUB_OUTPUT" - # A red weekly lane nobody must action is waste — upsert ONE tracking - # issue (never a new issue per week) with the reconciliation output, so - # failures have an owner-visible artifact with history in one place. - - name: Upsert tracking issue on failure - if: always() && (steps.reconcile.outputs.exit != '0' || steps.gate-reconcile.outputs.exit != '0' || needs.eval-slices.result != 'success' || needs.gate-census.result != 'success') + - name: Stamp trial history series + if: always() + run: | + for file in /tmp/paid-report/trial-outcomes.jsonl /tmp/gate-census-report/trial-outcomes.jsonl; do + if [ -f "$file" ]; then bun --no-install run scripts/eval-trial-series.ts "$file"; fi + done + + - name: Upload trial outcomes for pass-rate history + if: always() + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 + with: + name: trial-outcomes-periodic-a${{ github.run_attempt }} + path: | + /tmp/paid-report/trial-outcomes.jsonl + if-no-files-found: ignore + retention-days: 90 + + - name: Upload gate census trial outcomes for pass-rate history + if: always() + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 + with: + name: trial-outcomes-gate-census-a${{ github.run_attempt }} + path: | + /tmp/gate-census-report/trial-outcomes.jsonl + if-no-files-found: ignore + retention-days: 90 + + # Weekly pass-rate gate over the last 10 weekly runs (drift, rule cases + # behaving like behavior, quarantine exit/expiry/cap). Fails closed when + # history cannot be fetched. + - name: Pass-rate history gate + id: pass-rates + if: always() env: GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} + run: | + set +e + bun run eval:pass-rates --gate --runs 10 > /tmp/pass-rates.txt 2>&1 + echo "exit=$?" >> "$GITHUB_OUTPUT" + cat /tmp/pass-rates.txt + + # UC-E1 (approved): a run whose every red verdict is machine-classified + # INFRA or INCOMPLETE may be re-dispatched ONCE as a new run. + - name: Classify the census verdict + id: verdict + if: always() + env: + REDISPATCH_OF: ${{ inputs.redispatch_of }} + PERIODIC_EXIT: ${{ steps.reconcile.outputs.exit }} + GATE_EXIT: ${{ steps.gate-reconcile.outputs.exit }} + run: | + eligible() { # $1 exit, $2 report dir + [ "$1" = "0" ] && return 0 + jq -e '.version == 2 and .verdict.redispatchEligible == true' "$2/collector-outcomes.json" >/dev/null 2>&1 + } + if [ -z "$REDISPATCH_OF" ] && { [ "$PERIODIC_EXIT" != "0" ] || [ "$GATE_EXIT" != "0" ]; } \ + && eligible "$PERIODIC_EXIT" /tmp/paid-report && eligible "$GATE_EXIT" /tmp/gate-census-report; then + echo "redispatch=true" >> "$GITHUB_OUTPUT" + else + echo "redispatch=false" >> "$GITHUB_OUTPUT" + fi + + # A red weekly lane nobody must action is waste — upsert ONE tracking + # issue (never a new issue per week) with the headline and failure block + # of both lanes, and close it on the next green run. + - name: Upsert tracking issue on failure + if: always() && (steps.reconcile.outputs.exit != '0' || steps.gate-reconcile.outputs.exit != '0' || steps.pass-rates.outputs.exit != '0' || needs.eval-slices.result != 'success' || needs.gate-census.result != 'success') + env: + GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} + REDISPATCH: ${{ steps.verdict.outputs.redispatch }} + REDISPATCH_OF: ${{ inputs.redispatch_of }} run: | set -euo pipefail TITLE="Weekly periodic evals: red lane needs triage" BODY_FILE=/tmp/issue-body.md + RUN_URL="${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}/actions/runs/${GITHUB_RUN_ID}" { - echo "Automated weekly report — run: ${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}/actions/runs/${GITHUB_RUN_ID}" + echo "Automated weekly report — run: ${RUN_URL}" + if [ -n "$REDISPATCH_OF" ]; then echo; echo "This run is the one INFRA/INCOMPLETE re-dispatch of run ${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}/actions/runs/${REDISPATCH_OF}; both runs are reported."; fi + if [ "$REDISPATCH" = "true" ]; then echo; echo "Every red verdict is machine-classified INFRA/INCOMPLETE: re-dispatching once as a new run (EVAL_POLICY.infraRedispatch). This run stays red and reported."; fi echo - echo "- periodic reconciliation exit: ${{ steps.reconcile.outputs.exit }}" - echo "- periodic slices job: ${{ needs.eval-slices.result }}" - echo "- gate census reconciliation exit: ${{ steps.gate-reconcile.outputs.exit }}" - echo "- gate census job: ${{ needs.gate-census.result }}" + echo "- periodic reconciliation exit: ${{ steps.reconcile.outputs.exit }} (slices job: ${{ needs.eval-slices.result }})" + echo "- gate census reconciliation exit: ${{ steps.gate-reconcile.outputs.exit }} (census job: ${{ needs.gate-census.result }})" + echo "- pass-rate history gate exit: ${{ steps.pass-rates.outputs.exit }}" + echo + echo "### Periodic lane" + cat /tmp/paid-report/report-summary.md 2>/dev/null || echo "(no periodic report summary)" + echo + echo "### Gate census" + cat /tmp/gate-census-report/report-summary.md 2>/dev/null || echo "(no gate census report summary)" + echo + echo "### Pass-rate history (ACTION REQUIRED)" + echo '```' + { grep -E 'ACTION REQUIRED|history unavailable' /tmp/pass-rates.txt || echo "(no pass-rate alarms)"; } | sed 's/@/@\xe2\x80\x8b/g' | head -c 6000 + echo '```' + echo + echo "
Full reconciliation output" echo echo '```' - tail -c 6000 /tmp/report.txt 2>/dev/null || echo "(no reconciliation output)" + tail -c 6000 /tmp/report.txt 2>/dev/null | sed 's/@/@\xe2\x80\x8b/g' || echo "(no reconciliation output)" echo '```' echo echo '```' - tail -c 6000 /tmp/gate-report.txt 2>/dev/null || echo "(no gate census reconciliation output)" + tail -c 6000 /tmp/gate-report.txt 2>/dev/null | sed 's/@/@\xe2\x80\x8b/g' || echo "(no gate census reconciliation output)" echo '```' + echo "
" echo - echo "Exclusion policy: test/helpers/periodic-exclude-data.ts (every entry needs reason + tracking; removal re-activates the file next week)." + echo "Policy: EVAL_POLICY and CASE_QUARANTINE in test/helpers/periodic-exclude-data.ts; history: \`bun run eval:pass-rates\`." } > "$BODY_FILE" EXISTING=$(gh issue list --repo "$GITHUB_REPOSITORY" --state open --search "in:title \"$TITLE\"" --json number --jq '.[0].number // empty') if [ -n "$EXISTING" ]; then @@ -431,6 +524,37 @@ jobs: gh issue create --repo "$GITHUB_REPOSITORY" --title "$TITLE" --body-file "$BODY_FILE" fi + - name: Close the tracking issue on a green run + if: always() && steps.reconcile.outputs.exit == '0' && steps.gate-reconcile.outputs.exit == '0' && steps.pass-rates.outputs.exit == '0' && needs.eval-slices.result == 'success' && needs.gate-census.result == 'success' + env: + GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} + REDISPATCH_OF: ${{ inputs.redispatch_of }} + run: | + set -euo pipefail + TITLE="Weekly periodic evals: red lane needs triage" + EXISTING=$(gh issue list --repo "$GITHUB_REPOSITORY" --state open --search "in:title \"$TITLE\"" --json number --jq '.[0].number // empty') + if [ -n "$EXISTING" ]; then + NOTE="Green weekly run: ${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}/actions/runs/${GITHUB_RUN_ID}" + if [ -n "$REDISPATCH_OF" ]; then NOTE="${NOTE} (the INFRA re-dispatch of run ${REDISPATCH_OF}, which stays red and reported)"; fi + gh issue close "$EXISTING" --repo "$GITHUB_REPOSITORY" --comment "$NOTE" + fi + - name: Fail the workflow when reconciliation failed - if: always() && (steps.reconcile.outputs.exit != '0' || steps.gate-reconcile.outputs.exit != '0' || needs.eval-slices.result != 'success' || needs.gate-census.result != 'success') + if: always() && (steps.reconcile.outputs.exit != '0' || steps.gate-reconcile.outputs.exit != '0' || steps.pass-rates.outputs.exit != '0' || needs.eval-slices.result != 'success' || needs.gate-census.result != 'success') run: exit 1 + + # The one INFRA/INCOMPLETE re-dispatch (UC-E1). Its own job so the report + # job keeps no actions:write; the new run's concurrency group differs, so it + # never cancels this run. + redispatch: + runs-on: ubicloud-standard-2 + needs: report + if: ${{ !cancelled() && needs.report.outputs.redispatch == 'true' }} + timeout-minutes: 5 + permissions: + actions: write + steps: + - name: Re-dispatch the weekly census once + env: + GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} + run: gh workflow run evals-periodic.yml --repo "$GITHUB_REPOSITORY" --ref "$GITHUB_REF_NAME" -f redispatch_of="$GITHUB_RUN_ID" diff --git a/.github/workflows/evals.yml b/.github/workflows/evals.yml index 87775320b..7cf43e744 100644 --- a/.github/workflows/evals.yml +++ b/.github/workflows/evals.yml @@ -144,7 +144,7 @@ jobs: if: github.event_name != 'workflow_dispatch' || inputs.validation_phase == 'all' env: EVALS_ALL: ${{ (github.event_name == 'workflow_dispatch' && inputs.evals_all) && '1' || '' }} - run: EVALS_TIER=gate bun --no-install run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/paid-plan/manifest.json --slice-budget 540 --jobs 2 + run: EVALS_TIER=gate bun --no-install run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/paid-plan/manifest.json --slice-budget 540 --jobs 2 --max-parallel 16 - name: Emit validation-phase manifest if: github.event_name == 'workflow_dispatch' && inputs.validation_phase != 'all' @@ -299,11 +299,13 @@ jobs: path: /tmp/gstack-eval-input-cache key: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-${{ github.run_id }}-${{ github.run_attempt }}-${{ matrix.slice }} + # Attempt-scoped: a re-run attempt's trials are reported under that + # attempt and never replace (or collide with) the first attempt's. - name: Upload slice results if: always() uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 with: - name: paid-slice-${{ matrix.slice }} + name: paid-slice-${{ matrix.slice }}-a${{ github.run_attempt }} path: /tmp/paid-slice-results retention-days: 90 @@ -323,11 +325,13 @@ jobs: # The spooled per-shard full logs — a red weekly/PR lane three weeks # later needs more than a summary line. - - name: Upload shard logs on failure - if: failure() + # always(), not failure(): a failed behavior trial is a verdict and no + # longer reds its runner, but its full log is the diagnosis evidence. + - name: Upload shard logs + if: always() uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 with: - name: paid-slice-${{ matrix.slice }}-logs + name: paid-logs-slice-${{ matrix.slice }}-a${{ github.run_attempt }} include-hidden-files: true # The Fix-bun-temp step points TMPDIR at /home/runner/.cache, so the # runner's spool lands THERE, not /tmp — the original /tmp glob @@ -372,11 +376,13 @@ jobs: name: paid-plan path: /tmp/paid-report + # One directory per attempt-scoped slice artifact (no merge): shard + # records can never overwrite each other, and the report keeps the + # first attempt's verdict. - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8 with: - pattern: paid-slice-[0-9]* + pattern: paid-slice-* path: /tmp/paid-report - merge-multiple: true - name: Reconcile slices against the manifest (fail-closed) id: reconcile @@ -389,17 +395,31 @@ jobs: # (caught by the ship review army; the wiring test now pins this). echo "exit=${PIPESTATUS[0]}" >> "$GITHUB_OUTPUT" + - name: Stamp trial history series + if: always() && hashFiles('/tmp/paid-report/trial-outcomes.jsonl') != '' + run: bun --no-install run scripts/eval-trial-series.ts /tmp/paid-report/trial-outcomes.jsonl + - name: Upload reconciliation output for the comment job if: always() uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 with: - name: report-verdict + name: report-verdict-a${{ github.run_attempt }} path: | /tmp/report.txt /tmp/paid-report/collector-outcomes.json + /tmp/paid-report/report-summary.md if-no-files-found: ignore retention-days: 30 + - name: Upload trial outcomes for pass-rate history + if: always() + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 + with: + name: trial-outcomes-pr-a${{ github.run_attempt }} + path: /tmp/paid-report/trial-outcomes.jsonl + if-no-files-found: ignore + retention-days: 90 + - name: Fail the workflow when reconciliation failed if: steps.reconcile.outputs.exit != '0' run: exit 1 @@ -426,18 +446,13 @@ jobs: - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8 with: - pattern: paid-slice-[0-9]* - path: /tmp/paid-report - merge-multiple: true - - - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8 - with: - name: report-verdict + name: report-verdict-a${{ github.run_attempt }} path: /tmp/verdict continue-on-error: true - # Verified counts come from the read-only report job, not repo code in - # this write-token job. Keeps the + # Every count, verdict and failure line comes from the read-only report + # job's collector-outcomes v2 (panelVerdict() ran there); this job runs + # no repo code and never recomputes a verdict. Keeps the # "## E2E Evals" marker so the upsert keeps updating the same comment. # Runs even when reconciliation failed — a red lane on the PR is the point. - name: Post PR comment @@ -446,13 +461,14 @@ jobs: RECONCILE_EXIT: ${{ needs.slices-report.outputs.reconcile-exit }} run: | # shellcheck disable=SC2086,SC2059 - RESULTS=$(find /tmp/paid-report -name '*.json' ! -name 'manifest.json' ! -name 'slice-*.json' ! -name '_partial*' 2>/dev/null | sort) - TOTAL=0; PASSED=0; FAILED=0; MANUAL=0; FLAKY=0; EXECUTED=0; REUSED=0; COST="0" + TOTAL=0; PASSED=0; FAILED=0; MANUAL=0; EXECUTED=0; REUSED=0; COST="0" SUITE_LINES="" VERIFIED=/tmp/verdict/paid-report/collector-outcomes.json if ! jq -e ' . as $summary | - .version == 1 and (.files | type == "array") and (.totals | type == "object") and + .version == 2 and (.files | type == "array") and (.totals | type == "object") and + (.headline | type == "array") and (.failures | type == "array") and (.panels | type == "array") and + (.verdict.verdict == "GREEN" or .verdict.verdict == "RED") and ([.files[] | .total == (.passed + .failed + .manual_accepted) and (.total == (.executed + .reused)) and ([.total,.passed,.failed,.manual_accepted,.executed,.reused,.attempts,.flaky] | all(. >= 0 and (floor == .))) ] | all) and @@ -461,100 +477,68 @@ jobs: all(. as $key | ([$summary.files[] | .[$key]] | add // 0) == $summary.totals[$key])) ' "$VERIFIED" >/dev/null 2>&1; then VERIFIED="" - echo 'Verified collector summary unavailable; manual acceptance is unavailable/unverified.' + echo 'Verified report summary unavailable; no verdict, counts or manual acceptance can be shown.' fi + HEADLINE='(no verified report headline)' + FAILURES="" if [ -n "$VERIFIED" ]; then while IFS=$'\t' read -r f T P F M FL EX RE _ATTEMPTS C TIER SHARD; do [ "$T" -eq 0 ] && continue TOTAL=$((TOTAL + T)); PASSED=$((PASSED + P)); FAILED=$((FAILED + F)) - MANUAL=$((MANUAL + M)); FLAKY=$((FLAKY + FL)) + MANUAL=$((MANUAL + M)) EXECUTED=$((EXECUTED + EX)); REUSED=$((REUSED + RE)) COST=$(echo "$COST + $C" | bc) STATUS_ICON="✅" [ "$M" -gt 0 ] && STATUS_ICON="⚠ manual/unscored" [ "$F" -gt 0 ] && STATUS_ICON="❌" - [ "$F" -eq 0 ] && [ "$M" -eq 0 ] && [ "$FL" -gt 0 ] && STATUS_ICON="✅⚠" SUITE_LINES="${SUITE_LINES}| ${TIER}/${SHARD} | ${P}/${T} | ${M} | ${EX} | ${RE} | ${STATUS_ICON} | \$${C} |\n" done < <(jq -r '.files[] | [.file,.total,.passed,.failed,.manual_accepted,.flaky,.executed,.reused,.attempts,.cost,.tier,.shard] | @tsv' "$VERIFIED") - else - for f in $RESULTS; do - if ! jq -e '.total_tests' "$f" >/dev/null 2>&1; then - echo "Skipping malformed JSON: $f" - continue - fi - # FINAL-attempt accounting: eval-store keeps EVERY retry attempt - # as its own record (that's the flake telemetry), so counting raw - # records marks a pass-on-retry as a failure and inflates totals. - # Group by test name and judge the LAST record. Retry metadata - # includes both passing and failing final outcomes; show it separately. - # Guarded: a file with total_tests but a null/non-array `tests` - # passes the -e probe, the group_by then fails, and an empty $T - # would abort the whole step under bash -e ([ "" -eq 0 ] is an - # error) — killing the comment on exactly the corrupted-artifact - # runs where the red evidence matters (claude adversarial). - STATS=$(jq -r '[.tests | group_by(.name)[] | last] as $final | "\($final | length) \([$final[] | select(.passed)] | length) \([$final[] | select(.passed | not)] | length) \(.flaky_retries // [] | length) \([$final[] | select(.execution != "reused")] | length) \([$final[] | select(.execution == "reused")] | length)"' "$f" 2>/dev/null) || { echo "Skipping malformed tests[] in: $f"; continue; } - read -r T P F FL EX RE <<< "$STATS" - [ -z "$T" ] && { echo "Skipping malformed tests[] in: $f"; continue; } - C=$(jq -r '.total_cost_usd // 0' "$f") - TIER=$(jq -r '.tier // "unknown"' "$f") - SHARD=$(jq -r '.shard // "-"' "$f") - [ "$T" -eq 0 ] && continue - TOTAL=$((TOTAL + T)) - PASSED=$((PASSED + P)) - FAILED=$((FAILED + F)) - FLAKY=$((FLAKY + FL)) - EXECUTED=$((EXECUTED + EX)) - REUSED=$((REUSED + RE)) - COST=$(echo "$COST + $C" | bc) - STATUS_ICON="✅" - [ "$F" -gt 0 ] && STATUS_ICON="❌" - [ "$F" -eq 0 ] && [ "$FL" -gt 0 ] && STATUS_ICON="✅⚠" - SUITE_LINES="${SUITE_LINES}| ${TIER}/${SHARD} | ${P}/${T} | unverified | ${EX} | ${RE} | ${STATUS_ICON} | \$${C} |\n" - done + # Report-sanitized lines (no @-mentions, one capped line each), fenced here. + HEADLINE=$(jq -r '.headline[]' "$VERIFIED") + FAILURES=$(jq -r '.failures[]' "$VERIFIED") fi COVERAGE=$(jq -r '"Profile: \(.profile // "full") / \(.prCoverage.mode // "broad"); selected behaviors: \(.selection.e2e | if . == null then "all" else length end), judges: \(.selection.judges | if . == null then "all" else length end). Deferred to scheduled/release coverage: \(.prCoverage.deferred // [] | length) behaviors and \(.prCoverage.deferredPromptFiles // [] | length) changed prompt files. Deferred checks did not run and receive no PR-pass credit."' /tmp/paid-report/manifest.json) || COVERAGE='Coverage manifest unavailable; no coverage claim.' STATUS="✅ PASS" - if [ "${RECONCILE_EXIT:-1}" != "0" ] || [ "$FAILED" -gt 0 ]; then STATUS="❌ FAIL"; fi + if [ "${RECONCILE_EXIT:-1}" != "0" ]; then STATUS="❌ FAIL"; fi if [ "$STATUS" = '✅ PASS' ] && [ "$MANUAL" -gt 0 ]; then STATUS='⚠ MANUAL ACCEPTED (unscored)'; fi - if [ -z "$VERIFIED" ]; then STATUS='❌ FAIL (manual acceptance unavailable/unverified)'; fi + if [ -z "$VERIFIED" ]; then STATUS='❌ FAIL (verified report unavailable)'; fi BODY="## E2E Evals: ${STATUS} - **${PASSED} automated passed / ${TOTAL} final results** | **${FAILED} failed, ${MANUAL} manual accepted (unscored; no score-cache credit)** | **${EXECUTED} executed, ${REUSED} reused** | **\$${COST}** total cost | reconcile exit: ${RECONCILE_EXIT:-missing}$([ "$FLAKY" -gt 0 ] && printf ' | ⚠ %s cases with multiple attempts' "$FLAKY") + \`\`\` + ${HEADLINE} + \`\`\` + + **${EXECUTED} executed, ${REUSED} reused** rule/judge records | **${MANUAL} manual accepted (unscored; no score-cache credit)** | **\$${COST}** rule/judge cost | reconcile exit: ${RECONCILE_EXIT:-missing} ${COVERAGE} +
Rule and judge shards + | Shard | Automated result | Manual/unscored | Executed | Reused | Status | Cost | |-------|------------------|-----------------|----------|--------|--------|------| $(echo -e "$SUITE_LINES") +
Fail-closed reconciliation \`\`\` - $(tail -c 4000 /tmp/verdict/report.txt 2>/dev/null || echo '(no reconciliation output)') + $(tail -c 4000 /tmp/verdict/report.txt 2>/dev/null | sed 's/@/@\xe2\x80\x8b/g' || echo '(no reconciliation output)') \`\`\`
--- - *Sliced lane: declared PR profile or broad fallback via scripts/test-paid-shards.ts (planner → duration-packed executors → fail-closed report). Reused results retain their original provenance and expiry.*" + *Sliced lane: planner → duration-packed executors → fail-closed report. Behavior cases run a pre-registered 3-trial panel (PASS at 2/3 with no contract violation); a PASS 2/3 is shown with its failed trial, never as a clean pass. Reused results retain their original provenance and expiry.*" - if [ "$FAILED" -gt 0 ]; then - FAILURES="" - for f in $RESULTS; do - if ! jq -e '.failed' "$f" >/dev/null 2>&1; then continue; fi - if [ -n "$VERIFIED" ]; then - FAILS=$(jq -r '[.tests | group_by(.name)[] | last | select(.passed == false and (has("manual_review") | not))][] | "- ❌ \(.name): \(.exit_reason // "unknown")"' "$f" 2>/dev/null || echo "- ⚠️ parse error") - else - FAILS=$(jq -r '[.tests | group_by(.name)[] | last | select(.passed == false)][] | "- ❌ \(.name): \(.exit_reason // "unknown")"' "$f" 2>/dev/null || echo "- ⚠️ parse error") - fi - FAILURES="${FAILURES}${FAILS}\n" - done + if [ -n "$FAILURES" ]; then BODY="${BODY} - ### Failures - $(echo -e "$FAILURES")" + ### Failures and split verdicts + \`\`\` + ${FAILURES} + \`\`\`" fi COMMENT_ID=$(gh api repos/${{ github.repository }}/issues/${{ github.event.pull_request.number }}/comments \ diff --git a/test/ci-paid-coordination.test.ts b/test/ci-paid-coordination.test.ts index 5601e9087..42bea3025 100644 --- a/test/ci-paid-coordination.test.ts +++ b/test/ci-paid-coordination.test.ts @@ -117,10 +117,11 @@ describe('paid CI coordination stays off the eval image', () => { if (name === 'evals.yml') expect(report.permissions).toEqual({ contents: 'read' }); }); - test(`${name}: failure logs include the hidden spool directory without uploading the rest of the cache`, () => { - const logs = jobs['eval-slices'].steps.find(step => step.with?.name === 'paid-slice-${{ matrix.slice }}-logs'); + test(`${name}: shard logs include the hidden spool directory without uploading the rest of the cache`, () => { + const logs = jobs['eval-slices'].steps.find(step => step.with?.name === 'paid-logs-slice-${{ matrix.slice }}-a${{ github.run_attempt }}'); expect(logs?.uses).toStartWith('actions/upload-artifact@'); - expect(logs?.if).toBe('failure()'); + // A failed trial no longer reds its runner; its log is still the evidence. + expect(logs?.if).toBe('always()'); expect(logs?.with?.['include-hidden-files']).toBe(true); expect(String(logs?.with?.path).trim().split('\n')).toEqual([ '/home/runner/.cache/gstack-paid-shard-*.log', diff --git a/test/evals-workflow-wiring.test.ts b/test/evals-workflow-wiring.test.ts index 3b34a2783..485f10080 100644 --- a/test/evals-workflow-wiring.test.ts +++ b/test/evals-workflow-wiring.test.ts @@ -263,3 +263,68 @@ describe('shared setup composites (every paid lane)', () => { } }); }); + +describe('panel verdict surfaces (eval reliability policy)', () => { + type AnyJob = { if?: string; needs?: string[]; permissions?: Record; outputs?: Record; + strategy?: { 'max-parallel': number }; steps: Array }; + const jobsOf = (source: string) => (Bun.YAML.parse(source) as { jobs: Record }).jobs; + + test('planners size the capacity preflight with their executor cap', () => { + for (const [source, executor, manifest] of [[evalsYml, 'eval-slices', '/tmp/paid-plan/manifest.json'], + [periodicYml, 'eval-slices', '/tmp/paid-plan/manifest.json'], [periodicYml, 'gate-census', '/tmp/gate-census-plan/manifest.json']] as const) { + const jobs = jobsOf(source); + const emit = jobs['plan-slices']!.steps.find(step => step.run?.includes(`--emit-plan ${manifest} `))!; + const cap = Number(/--max-parallel (\d+)/.exec(emit.run!)?.[1]); + expect(cap, `${executor}: --max-parallel`).toBe(jobs[executor]!.strategy!['max-parallel']); + } + }); + + test('slice artifacts are attempt-scoped and never merged into one tree', () => { + for (const source of [evalsYml, periodicYml, marathonYml]) { + const jobs = jobsOf(source); + const uploads = Object.values(jobs).flatMap(job => job.steps).filter(step => step.uses?.startsWith('actions/upload-artifact@')) + .map(step => step.with?.name ?? '').filter(name => /slice|census-\$/.test(name)); + expect(uploads.length).toBeGreaterThan(0); + for (const name of uploads) expect(name, name).toContain('-a${{ github.run_attempt }}'); + const downloads = Object.values(jobs).flatMap(job => job.steps).filter(step => step.uses?.startsWith('actions/download-artifact@') && step.with?.pattern); + for (const step of downloads) expect((step.with as Record)['merge-multiple'], step.with!.pattern).toBeUndefined(); + } + }); + + test('the PR comment reads collector-outcomes v2 and never recomputes a verdict', () => { + const comment = evalsYml.slice(evalsYml.indexOf(' slices-comment:')); + expect(comment).toContain('.version == 2'); + expect(comment).toContain("jq -r '.failures[]'"); + expect(comment).toContain('name: report-verdict-a${{ github.run_attempt }}'); + expect(evalsYml).not.toContain('group_by(.name)'); + expect(comment).not.toMatch(/paid-slice-/); + const report = jobsOf(evalsYml)['slices-report']!; + expect(report.steps.some(step => step.run?.includes('scripts/eval-trial-series.ts /tmp/paid-report/trial-outcomes.jsonl'))).toBe(true); + expect(report.steps.some(step => step.with?.name?.startsWith('trial-outcomes-'))).toBe(true); + }); + + test('the weekly report gates on pass-rate history, closes its issue on green, and re-dispatches INFRA-only reds once', () => { + const jobs = jobsOf(periodicYml); + const report = jobs.report!; + expect(report.permissions).toEqual({ contents: 'read', issues: 'write', actions: 'read' }); + const gate = report.steps.find(step => step.id === 'pass-rates')!; + expect(gate.run).toContain('bun run eval:pass-rates --gate --runs 10'); + expect(gate.if).toBe('always()'); + for (const name of ['Upsert tracking issue on failure', 'Fail the workflow when reconciliation failed']) { + expect(report.steps.find(step => step.name === name)!.if).toContain("steps.pass-rates.outputs.exit != '0'"); + } + const upsert = report.steps.find(step => step.name === 'Upsert tracking issue on failure')!; + expect(upsert.run).toContain('report-summary.md'); + expect(report.steps.find(step => step.name === 'Close the tracking issue on a green run')!.run).toContain('gh issue close'); + expect(report.steps.filter(step => step.with?.name?.startsWith('trial-outcomes-')).length).toBe(2); + const redispatch = jobs.redispatch!; + expect([redispatch.needs].flat()).toEqual(['report']); + expect(redispatch.permissions).toEqual({ actions: 'write' }); + expect(redispatch.if).toBe("${{ !cancelled() && needs.report.outputs.redispatch == 'true' }}"); + expect(redispatch.steps[0]!.run).toContain('-f redispatch_of="$GITHUB_RUN_ID"'); + const classify = report.steps.find(step => step.id === 'verdict')!; + expect(classify.run).toContain('.verdict.redispatchEligible == true'); + expect(classify.run).toContain('[ -z "$REDISPATCH_OF" ]'); + expect(periodicYml).toMatch(/group: evals-periodic\$\{\{ inputs\.redispatch_of/); + }); +}); From 3bae8e33da5beb9b1cb118ca92ed1e98e6682340 Mon Sep 17 00:00:00 2001 From: garrytan Date: Tue, 29 Sep 2026 19:41:39 +0000 Subject: [PATCH 14/15] feat(evals): planner-side whole-panel reuse and negative receipts The planner job restores this PR's receipt store once and ships a single filtered set with the plan: a pass or panel receipt with a same-or-newer FAIL for its input identity is dropped, and a panel receipt ships only as a whole PASS panel (re-verified with panelVerdict) from one run. Executors read only that set (no per-slice cache restore or save), so every trial of a panel sees the same receipts; a trial reuses its own record from the panel receipt, keeping a split PASS's failed trial. Trial identities drop the trial index (run-scoped) and bind the panel policy. Executed shards carry their input identity; the report turns a whole fresh PASS panel into a panel receipt and a FAIL panel or failed rule shard into a negative receipt, and marks a panel that mixes reused and fresh trials INCOMPLETE. The report job merges plan, slice and report receipts (newest per file) and saves one store per run. Also fixes two TS2352 casts in browse/test/dia-macos-qualification.test.ts whose diagnostic text drifted with program order (baseline locked, fix only). --- .github/workflows/evals.yml | 84 ++++++----- browse/test/dia-macos-qualification.test.ts | 4 +- scripts/e2e-shard-reuse.ts | 149 +++++++++++++++++++- scripts/test-paid-shards.ts | 73 ++++++++-- scripts/typecheck-test-baseline.json | 1 - test/ci-eval-cache.test.ts | 98 ++++++------- test/e2e-shard-reuse.test.ts | 78 +++++++++- test/paid-report-fail-open.test.ts | 19 ++- test/paid-run-manifest.test.ts | 14 ++ 9 files changed, 410 insertions(+), 110 deletions(-) diff --git a/.github/workflows/evals.yml b/.github/workflows/evals.yml index 7cf43e744..8dbbc99bd 100644 --- a/.github/workflows/evals.yml +++ b/.github/workflows/evals.yml @@ -140,10 +140,23 @@ jobs: with: bun-version: 1.4.0 + # Planner-side reuse: restore this PR's newest receipt store (the report + # job saves one merged store per run) and ship ONE filtered set with the + # plan, so every trial of a panel sees the same receipts and a newer FAIL + # blocks any older PASS for the same inputs. + - name: Restore this PR's verified judge and E2E results + if: github.event_name == 'pull_request' + uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6 + with: + path: /tmp/gstack-eval-input-cache + key: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-${{ github.run_id }}-${{ github.run_attempt }}-plan + restore-keys: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}- + - name: Emit run manifest if: github.event_name != 'workflow_dispatch' || inputs.validation_phase == 'all' env: EVALS_ALL: ${{ (github.event_name == 'workflow_dispatch' && inputs.evals_all) && '1' || '' }} + EVALS_CACHE_DIR: ${{ github.event_name == 'pull_request' && '/tmp/gstack-eval-input-cache' || '' }} run: EVALS_TIER=gate bun --no-install run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/paid-plan/manifest.json --slice-budget 540 --jobs 2 --max-parallel 16 - name: Emit validation-phase manifest @@ -182,7 +195,9 @@ jobs: - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 with: name: paid-plan - path: /tmp/paid-plan/manifest.json + path: | + /tmp/paid-plan/manifest.json + /tmp/paid-plan/receipts retention-days: 30 eval-slices: @@ -251,15 +266,13 @@ jobs: name: paid-plan path: /tmp/paid-plan - # Only this PR's receipts are eligible. No base-branch or cross-PR restore - # prefix; every receipt also verifies exact inputs and its original age. - - name: Restore this PR's verified judge and E2E results - if: github.event_name == 'pull_request' - uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6 - with: - path: /tmp/gstack-eval-input-cache - key: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-${{ github.run_id }}-${{ github.run_attempt }}-${{ matrix.slice }} - restore-keys: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}- + # Receipts come only from the plan (this PR's store, filtered once by the + # planner); new receipts land beside the slice results and the report + # merges them into the next store. + - name: Seed this slice's receipts from the plan + run: | + mkdir -p /tmp/paid-slice-results/receipts + if [ -d /tmp/paid-plan/receipts ]; then cp -a /tmp/paid-plan/receipts/. /tmp/paid-slice-results/receipts/; fi - name: Run slice ${{ matrix.slice }} env: @@ -270,35 +283,12 @@ jobs: EVALS_JOBS: "2" EVALS_CONCURRENCY: "2" GSTACK_EVAL_DIR: /tmp/paid-slice-results - EVALS_CACHE_DIR: /tmp/gstack-eval-input-cache + EVALS_CACHE_DIR: /tmp/paid-slice-results/receipts EVALS_CACHE_REPOSITORY: ${{ github.repository }} EVALS_CACHE_PR: ${{ github.event.pull_request.number }} EVALS_CACHE_RUNTIME_ID: ${{ needs.build-image.outputs.runtime-id }} run: EVALS_TIER=gate bun run scripts/test-paid-shards.ts --tier gate --plan /tmp/paid-plan/manifest.json --slice ${{ matrix.slice }} - - name: Find finalized passing receipts - id: receipts - if: ${{ !cancelled() && github.event_name == 'pull_request' }} - run: | - # Only a producer publishes. A later reuse-only slice must not become - # the newest prefix match and hide another slice's newly earned pass. - for receipt in /tmp/gstack-eval-input-cache/*.json; do - [ -f "$receipt" ] || continue - if jq -e --arg run "$GITHUB_RUN_ID/$GITHUB_RUN_ATTEMPT" '.proof.source.runId == $run' "$receipt" >/dev/null 2>&1; then - echo 'present=true' >> "$GITHUB_OUTPUT" - break - fi - done - - # An unrelated failing case does not discard already verified passes. - # Failed/retried/partial attempts never become receipts in the first place. - - name: Save verified judge and E2E results for this PR - if: ${{ !cancelled() && steps.receipts.outputs.present == 'true' }} - uses: actions/cache/save@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6 - with: - path: /tmp/gstack-eval-input-cache - key: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-${{ github.run_id }}-${{ github.run_attempt }}-${{ matrix.slice }} - # Attempt-scoped: a re-run attempt's trials are reported under that # attempt and never replace (or collide with) the first attempt's. - name: Upload slice results @@ -396,8 +386,27 @@ jobs: echo "exit=${PIPESTATUS[0]}" >> "$GITHUB_OUTPUT" - name: Stamp trial history series - if: always() && hashFiles('/tmp/paid-report/trial-outcomes.jsonl') != '' - run: bun --no-install run scripts/eval-trial-series.ts /tmp/paid-report/trial-outcomes.jsonl + if: always() + run: | + if [ -f /tmp/paid-report/trial-outcomes.jsonl ]; then + bun --no-install run scripts/eval-trial-series.ts /tmp/paid-report/trial-outcomes.jsonl + fi + + # One merged receipt store per run: the plan's shipped set, every slice's + # new pass receipts, and the report's panel and negative receipts. Saved + # last, so the next planner restores it as the newest prefix match. + - name: Merge this run's receipts + if: always() && github.event_name == 'pull_request' + run: | + bun --no-install run scripts/e2e-shard-reuse.ts merge /tmp/gstack-eval-input-cache \ + /tmp/paid-report/receipts /tmp/paid-report/report-receipts /tmp/paid-report/paid-slice-*/receipts + + - name: Save this PR's verified judge and E2E results + if: always() && github.event_name == 'pull_request' + uses: actions/cache/save@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6 + with: + path: /tmp/gstack-eval-input-cache + key: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-${{ github.run_id }}-${{ github.run_attempt }}-merged - name: Upload reconciliation output for the comment job if: always() @@ -501,7 +510,8 @@ jobs: COVERAGE=$(jq -r '"Profile: \(.profile // "full") / \(.prCoverage.mode // "broad"); selected behaviors: \(.selection.e2e | if . == null then "all" else length end), judges: \(.selection.judges | if . == null then "all" else length end). Deferred to scheduled/release coverage: \(.prCoverage.deferred // [] | length) behaviors and \(.prCoverage.deferredPromptFiles // [] | length) changed prompt files. Deferred checks did not run and receive no PR-pass credit."' /tmp/paid-report/manifest.json) || COVERAGE='Coverage manifest unavailable; no coverage claim.' STATUS="✅ PASS" - if [ "${RECONCILE_EXIT:-1}" != "0" ]; then STATUS="❌ FAIL"; fi + if [ "${RECONCILE_EXIT:-1}" != "0" ] || [ "$FAILED" -gt 0 ] \ + || { [ -n "$VERIFIED" ] && [ "$(jq -r '.verdict.verdict' "$VERIFIED")" != "GREEN" ]; }; then STATUS="❌ FAIL"; fi if [ "$STATUS" = '✅ PASS' ] && [ "$MANUAL" -gt 0 ]; then STATUS='⚠ MANUAL ACCEPTED (unscored)'; fi if [ -z "$VERIFIED" ]; then STATUS='❌ FAIL (verified report unavailable)'; fi diff --git a/browse/test/dia-macos-qualification.test.ts b/browse/test/dia-macos-qualification.test.ts index 40e13bf67..48e483158 100644 --- a/browse/test/dia-macos-qualification.test.ts +++ b/browse/test/dia-macos-qualification.test.ts @@ -1214,7 +1214,7 @@ Binary Images: { status: 1, stdout: '', stderr: '' }, { status: 0, stdout: '', stderr: '' }, { status: 0, stdout: 'truncated-private-row', stderr: '' }, { status: null, stdout: null, stderr: null, error: new Error('synthetic-private-error') }, - ]) expect(inspectUidProcesses(23456, performance.now() + 10_000, {}, (() => result) as typeof spawnSync)).toEqual({ available: false }); + ]) expect(inspectUidProcesses(23456, performance.now() + 10_000, {}, (() => result) as unknown as typeof spawnSync)).toEqual({ available: false }); }); test('numeric UID process filtering runs through the real global process table', () => { @@ -1251,7 +1251,7 @@ Binary Images: { status: 113, stdout: '', stderr: 'Could not find domain for user uid: 23456' }, { status: null, stdout: null, stderr: null, error: new Error('synthetic-private-error') }, ]) { - const observation = inspectUserDomain(23456, performance.now() + 10_000, {}, (() => result) as typeof spawnSync); + const observation = inspectUserDomain(23456, performance.now() + 10_000, {}, (() => result) as unknown as typeof spawnSync); expect(observation.state).toBe('unavailable'); expect(observation.structure).toBeUndefined(); expect(JSON.stringify(observation)).not.toContain('synthetic-private'); diff --git a/scripts/e2e-shard-reuse.ts b/scripts/e2e-shard-reuse.ts index 692d89250..f0a905a65 100644 --- a/scripts/e2e-shard-reuse.ts +++ b/scripts/e2e-shard-reuse.ts @@ -30,6 +30,8 @@ import { buildEvalInputIdentity, lookupEvalInputCache, sourceDependencyClosure, type EvalCacheValue, type EvalInputIdentity, type EvalPassingProof } from './eval-input-cache'; import { matchGlob } from '../test/helpers/test-selection'; import { E2E_TOUCHFILES, GLOBAL_TOUCHFILES } from '../test/helpers/touchfiles-data'; +import { EVAL_CACHE_MAX_AGE_MS as RECEIPT_MAX_AGE_MS } from './eval-input-cache'; +import { panelVerdict, TRIAL_ENV, type EvalCaseKind, type PanelShape, type PanelTrial } from '../test/helpers/eval-store'; export interface E2EShardReuseRequest { root: string; @@ -50,6 +52,8 @@ export interface E2EShardReuseRequest { profile: string; /** The exact environment the child receives. */ env: NodeJS.ProcessEnv; + /** Isolated trial shard: its panel policy is part of the identity; the trial index is run-scoped. */ + panel?: { kind: EvalCaseKind; panel: PanelShape; quarantined: boolean }; } export interface E2EShardReuseHit { key: string; source: EvalPassingProof['source'] } @@ -61,7 +65,7 @@ const HARNESS_FILES = ['scripts/test-paid-shards.ts', 'scripts/e2e-shard-reuse.t const ENV_PREFIXES = ['EVALS_', 'GSTACK_', 'CLAUDE_', 'ANTHROPIC_', 'OPENAI_', 'GEMINI_', 'BUN_', 'NODE_', 'PLAYWRIGHT_']; /** Run-scoped values: provenance or transport, never behavior. Selection is bound as case ids. */ const RUN_SCOPED_ENV = new Set(['EVALS_RUN_ID', 'GSTACK_EVAL_DIR', 'EVALS_CACHE_DIR', 'EVALS_CACHE_PR', 'EVALS_CACHE_REPOSITORY', - 'EVALS_CACHE_RUNTIME_ID', 'EVALS_CACHE_PURPOSE', 'EVALS_SELECTION_JSON', 'EVALS_JUDGE_SELECTION_JSON']); + 'EVALS_CACHE_RUNTIME_ID', 'EVALS_CACHE_PURPOSE', 'EVALS_SELECTION_JSON', 'EVALS_JUDGE_SELECTION_JSON', TRIAL_ENV.trial]); const SECRET_ENV = /KEY|TOKEN|SECRET|PASSWORD|CREDENTIAL/; /** The reuse-relevant environment the child sees; secrets contribute presence only. */ @@ -126,7 +130,8 @@ export function e2eShardIdentity(request: E2EShardReuseRequest): { status: 'elig coverage: { dependencies: 'complete', prompts: 'complete', environment: 'complete' }, unknownDependencies: [], files, prompts: Object.fromEntries(request.caseIds.map(id => [id, source])), - parameters: { rootPackage, key: request.key, caseIds: [...request.caseIds].sort(), casePattern: request.casePattern, + parameters: { rootPackage, key: request.panel ? request.key.replace(/~t\d+$/, '') : request.key, + ...(request.panel ? { panel: { kind: request.panel.kind, n: request.panel.panel.n, k: request.panel.panel.k, quarantined: request.panel.quarantined } } : {}), caseIds: [...request.caseIds].sort(), casePattern: request.casePattern, expectedCases: request.expectedCases, retries: request.retries, timeoutMs: request.timeoutMs, withinShardConcurrency: request.withinShardConcurrency, tier: request.tier, profile: request.profile, environment: e2eReuseEnvironment(env) }, @@ -156,7 +161,13 @@ const validResult = (identity: EvalInputIdentity, key: string) => (value: EvalCa * whose inputs did not change during execution. */ export function prepareE2EShardReuse(request: E2EShardReuseRequest): { + /** The input identity key: recorded on the outcome so the report can store verdicts against it. */ + inputKey: string; lookup(): E2EShardReuseHit | null; + /** Trial shards only: this trial's record from a whole PASS panel receipt of the plan's receipts. */ + lookupPanelTrial(trial: number): { hit: E2EShardReuseHit; trial: PanelTrial } | null; + /** True when the inputs are unchanged since `before` (the outcome may carry inputKey). */ + unchanged(): boolean; publish(): void; } | null { if (e2eReuseLaneProblem(request.env, 'pr') !== null) return null; @@ -164,6 +175,19 @@ export function prepareE2EShardReuse(request: E2EShardReuseRequest): { if (before.status !== 'eligible') return null; const common = { cacheDir: request.env.EVALS_CACHE_DIR!, purpose: 'gate' as const }; return { + inputKey: before.identity.key, + unchanged() { + const after = e2eShardIdentity(request); + return after.status === 'eligible' && after.identity.key === before.identity.key; + }, + lookupPanelTrial(trial) { + if (!request.panel) return null; + const receipt = readPanelReceipt(common.cacheDir, before.identity.key); + if (!receipt || receipt.case !== request.caseIds[0] || receipt.kind !== request.panel.kind + || receipt.panel.n !== request.panel.panel.n || receipt.panel.k !== request.panel.panel.k) return null; + const record = receipt.trials.find(t => t.trial === trial); + return record ? { hit: { key: receipt.key, source: receipt.source }, trial: record } : null; + }, lookup() { const found = lookupEvalInputCache({ ...common, identity: before.identity, validateResult: validResult(before.identity, request.key) }); return found.status === 'reused' ? { key: found.key, source: found.source } : null; @@ -184,3 +208,124 @@ export function prepareE2EShardReuse(request: E2EShardReuseRequest): { }, }; } + +// ─── Panel receipts, negative receipts and the planner's receipt selection ── +// +// Reuse is decided by the planner, once per panel: it ships the plan a +// receipt set in which every panel receipt is a whole PASS panel from one run +// and no pass receipt has a newer FAIL for the same identity. Executors look +// up only that set, so every trial of a panel sees the same receipts. The +// report writes panel receipts (all n trials fresh, one identity) and +// negative receipts (FAIL verdicts) after the verdict is known. + +export interface PanelReceipt { + schema: 1; + key: string; + case: string; + kind: EvalCaseKind; + panel: PanelShape; + trials: PanelTrial[]; + source: { runId: string; revision: string; completedAt: number }; +} + +export interface NegativeReceipt { schema: 1; key: string; source: { runId: string; revision: string; completedAt: number } } + +const RECEIPT_KEY = /^[a-f0-9]{64}$/; +const validSource = (source: any) => !!source && typeof source.runId === 'string' && /^[\w./-]{1,160}$/.test(source.runId) + && typeof source.revision === 'string' && /^[a-f0-9]{40}$/.test(source.revision) && Number.isSafeInteger(source.completedAt) && source.completedAt > 0; + +function readJson(file: string, maxBytes = 64 * 1024): any { + try { + const stat = fs.lstatSync(file); + if (!stat.isFile() || stat.size > maxBytes) return null; + return JSON.parse(fs.readFileSync(file, 'utf8')); + } catch { return null; } +} + +/** A whole, unexpired PASS panel receipt for `key`, re-verified with panelVerdict(); else null. */ +export function readPanelReceipt(cacheDir: string, key: string, now = Date.now()): PanelReceipt | null { + if (!RECEIPT_KEY.test(key)) return null; + const receipt = readJson(path.join(cacheDir, `${key}.panel.json`)); + if (!receipt || receipt.schema !== 1 || receipt.key !== key || typeof receipt.case !== 'string' || !validSource(receipt.source) + || receipt.source.completedAt > now || now - receipt.source.completedAt >= RECEIPT_MAX_AGE_MS || !Array.isArray(receipt.trials)) return null; + try { + const verdict = panelVerdict({ case: receipt.case, kind: receipt.kind, panel: receipt.panel, + trials: receipt.trials.map((t: PanelTrial) => ({ ...t, attempt: 1 })) }); + if (verdict.status !== 'PASS' || verdict.trials.length !== receipt.panel.n) return null; + } catch { return null; } + const negative = readJson(path.join(cacheDir, `${key}.fail.json`)); + if (negative && validSource(negative.source) && negative.source.completedAt >= receipt.source.completedAt) return null; + return receipt as PanelReceipt; +} + +export function writePanelReceipt(dir: string, receipt: PanelReceipt): void { + fs.mkdirSync(dir, { recursive: true }); + fs.writeFileSync(path.join(dir, `${receipt.key}.panel.json`), `${JSON.stringify(receipt)}\n`, { mode: 0o600 }); +} + +export function writeNegativeReceipt(dir: string, receipt: NegativeReceipt): void { + fs.mkdirSync(dir, { recursive: true }); + fs.writeFileSync(path.join(dir, `${receipt.key}.fail.json`), `${JSON.stringify(receipt)}\n`, { mode: 0o600 }); +} + +const receiptTime = (file: string): number => { + const parsed = readJson(file); + return Number(parsed?.source?.completedAt ?? parsed?.proof?.source?.completedAt) || 0; +}; + +/** + * Planner-side selection: copy `from` into `to`, dropping every pass or panel + * receipt that has a same-or-newer negative receipt for its identity, and + * every panel receipt that is not a whole PASS panel. Workflow-judge and + * other receipts pass through for their own validation at lookup. + */ +export function selectPlanReceipts(from: string, to: string, now = Date.now()): { shipped: number; blocked: string[] } { + fs.mkdirSync(to, { recursive: true }); + const blocked: string[] = []; + let shipped = 0; + let names: string[] = []; + try { names = fs.readdirSync(from).filter(name => name.endsWith('.json')); } catch { return { shipped, blocked }; } + for (const name of names) { + const file = path.join(from, name); + const [key, suffix] = [name.slice(0, 64), name.slice(64)]; + const negative = RECEIPT_KEY.test(key) ? readJson(path.join(from, `${key}.fail.json`)) : null; + const newerFail = negative && validSource(negative.source) && negative.source.completedAt >= receiptTime(file); + if (suffix === '.panel.json' && (newerFail || !readPanelReceipt(from, key, now))) { blocked.push(name); continue; } + if (suffix === '.json' && newerFail) { blocked.push(name); continue; } + fs.copyFileSync(file, path.join(to, name)); + shipped++; + } + return { shipped, blocked }; +} + +/** Merge receipt directories into one store, keeping the newest file per name. */ +export function mergeReceiptDirs(out: string, dirs: string[]): number { + fs.mkdirSync(out, { recursive: true }); + let merged = 0; + for (const dir of dirs) { + let names: string[] = []; + try { names = fs.readdirSync(dir).filter(name => name.endsWith('.json')); } catch { continue; } + for (const name of names) { + const source = path.join(dir, name); + const target = path.join(out, name); + if (!fs.lstatSync(source).isFile()) continue; + if (fs.existsSync(target) && receiptTime(target) >= receiptTime(source)) continue; + fs.copyFileSync(source, target); + merged++; + } + } + return merged; +} + +if (import.meta.main) { + const [command, first, ...rest] = process.argv.slice(2); + if (command === 'select' && first && rest[0]) { + const result = selectPlanReceipts(first, rest[0]); + console.log(`[e2e-reuse] shipped ${result.shipped} receipt(s) to the plan; blocked ${result.blocked.length} (newer FAIL or partial panel)`); + } else if (command === 'merge' && first) { + console.log(`[e2e-reuse] merged ${mergeReceiptDirs(first, rest)} receipt(s) into ${first}`); + } else { + console.error('usage: bun run scripts/e2e-shard-reuse.ts select | merge '); + process.exit(2); + } +} diff --git a/scripts/test-paid-shards.ts b/scripts/test-paid-shards.ts index 198f8aceb..1a5d9c8b4 100644 --- a/scripts/test-paid-shards.ts +++ b/scripts/test-paid-shards.ts @@ -78,7 +78,7 @@ import { manualReviewProblem } from '../test/helpers/cookie-workflow-manual-revi import { preflightAnthropicApi } from '../test/helpers/anthropic-preflight'; import { OVERLAY_MIN_FILE_WALL_MS } from '../test/helpers/overlay-case-policy'; import { PR_PROFILE_CASE_IDS, PR_PROFILE_FILES, packageChangeOnlyVersion, selectPrProfile, type PrProfileSelection } from './test-pr-profile'; -import { e2eReuseLaneProblem, prepareE2EShardReuse } from './e2e-shard-reuse'; +import { e2eReuseLaneProblem, prepareE2EShardReuse, selectPlanReceipts, writeNegativeReceipt, writePanelReceipt } from './e2e-shard-reuse'; type E2EShardReuse = NonNullable>; import { @@ -826,6 +826,8 @@ export interface ShardOutcome { reused?: { inputKey: string; runId: string; revision: string; completedAt: number }; /** The parent could not run the shard at all (a runner error, never a trial verdict). */ runnerError?: string; + /** PR lane: the reuse input identity of a freshly executed shard whose inputs stayed unchanged. */ + inputKey?: string; /** Isolated trial shards only: the trial record this shard produced. */ trial?: ShardTrialRecord; } @@ -1057,7 +1059,22 @@ export async function runPaidShard( // Bootstrap-retention qualification binds per-run state, so that shard stays fresh. const reuse = files.some(file => normalizeRelativePath(file) === 'test/skill-e2e-qa-workflow.test.ts') ? null : options.reuseFor?.(files, env, budget) ?? null; - const reused = reuse?.lookup() ?? null; + // A trial reuses only its record from a whole PASS panel receipt the + // planner shipped; a single trial never has a pass receipt of its own. + const panelHit = trialPlan && trialIndex !== null ? reuse?.lookupPanelTrial(trialIndex) ?? null : null; + if (panelHit && trialPlan && trialIndex !== null && caseId !== null) { + const reusedFrom = { inputKey: panelHit.hit.key, runId: panelHit.hit.source.runId, revision: panelHit.hit.source.revision, completedAt: panelHit.hit.source.completedAt }; + const passedTrial = panelHit.trial.outcome === 'passed'; + log(`${label} REUSED trial ${trialIndex}/${trialPlan.panel.n} of ${caseId} (${panelHit.trial.outcome}) from the whole PASS panel of run ${reusedFrom.runId}`); + return { shard: shardNumber, files, status: passedTrial ? 'passed' : 'failed', exitCode: passedTrial ? 0 : 1, elapsedMs: 0, groupPid: null, + executedTests: 1, skippedTests: 0, budget, reused: reusedFrom, + trial: { case: caseId, trial: trialIndex, kind: trialPlan.kind, panel: trialPlan.panel, quarantined: trialPlan.quarantined, + outcome: panelHit.trial.outcome, cost_usd: 0, duration_ms: 0, + ...(panelHit.trial.failure_class ? { failure_class: panelHit.trial.failure_class } : {}), + ...(panelHit.trial.exit_reason ? { exit_reason: panelHit.trial.exit_reason } : {}), + ...(panelHit.trial.error ? { error: panelHit.trial.error } : {}) } }; + } + const reused = trialPlan ? null : reuse?.lookup() ?? null; if (reused) { const reusedFrom = { input_key: reused.key, run_id: reused.source.runId, revision: reused.source.revision, completed_at: new Date(reused.source.completedAt).toISOString() }; @@ -1255,7 +1272,8 @@ export async function runPaidShard( } } const elapsedMs = Date.now() - startedAt; - if (status === 'passed' && reuse) reuse.publish(); + if (status === 'passed' && reuse && !trialPlan) reuse.publish(); + const inputKey = reuse?.unchanged() ? reuse.inputKey : undefined; // Failure debuggability without the RAM cost: read back only the log's // tail. Live mode already streamed everything, so no re-print there. @@ -1273,7 +1291,8 @@ export async function runPaidShard( ? summary.terminalTestCounts.reduce((a, b) => a + b, 0) : null; const skippedTests = summary.terminalTestCounts.length > 0 ? summary.skippedTests : null; - return withTrial({ shard: shardNumber, files, status, exitCode, elapsedMs, groupPid, executedTests, skippedTests, budget }); + return withTrial({ shard: shardNumber, files, status, exitCode, elapsedMs, groupPid, executedTests, skippedTests, budget, + ...(inputKey ? { inputKey } : {}) }); } export interface RunSummary { @@ -2027,7 +2046,7 @@ export interface SliceResult { /** Epoch ms bounds of the slice's shard execution (lane wall time). */ startedAt?: number; finishedAt?: number; - outcomes: Array>; + outcomes: Array>; } /** @@ -2113,11 +2132,13 @@ export function verifySliceResults( if (outcome.reused !== undefined) { const r = outcome.reused; if (manifest.prCoverage?.mode !== 'pr') problems.push(`${file}: only the fast PR profile may reuse results; this lane executes fresh`); - if (outcome.status !== 'passed' || outcome.exitCode !== 0 || !/^[a-f0-9]{64}$/.test(r?.inputKey ?? '') + const reusedVerdictOk = outcome.trial ? outcome.trial.outcome !== null : outcome.status === 'passed' && outcome.exitCode === 0; + if (!reusedVerdictOk || !/^[a-f0-9]{64}$/.test(r?.inputKey ?? '') || !/^[\w./-]{1,160}$/.test(r?.runId ?? '') || !/^[a-f0-9]{40}$/.test(r?.revision ?? '') || !Number.isSafeInteger(r?.completedAt) || r.completedAt <= 0) { problems.push(`${file}: malformed reused result`); } } + if (outcome.inputKey !== undefined && !/^[a-f0-9]{64}$/.test(outcome.inputKey)) problems.push(`${file}: malformed input identity`); if (reported.has(file)) problems.push(`${file} reported by two slices`); reported.set(file, { slice: result.sliceIndex, status: outcome.status, ...(outcome.trial ? { trial: outcome.trial } : {}) }); const registered = FILE_RETRY_BUDGETS.find(budget => budget.file === shardFile(file)); @@ -2301,6 +2322,12 @@ export function panelReports(manifest: PaidRunManifest, results: SliceResult[], ...(t.timeout_at_turn !== undefined ? { timeout_at_turn: t.timeout_at_turn } : {}) }]; }); const verdict = panelVerdict({ case: shardCaseId(key)!, kind: plan.kind, panel: plan.panel, trials, quarantined: plan.quarantined }); + // Reuse is whole-panel only: every trial reused from one run, or none. + const sources = entries.map(entry => reported.get(normalizeRelativePath(entry.file))?.outcome.reused?.runId ?? null); + if (sources.some(source => source !== null) && (sources.some(source => source === null) || new Set(sources).size > 1)) { + return { ...verdict, status: 'INCOMPLETE', split: false, failsLane: true, coverage: false, redClass: 'INCOMPLETE', + reason: 'partial panel reuse (every trial must come from one reused panel, or none)', file: shardFile(key), slices }; + } return { ...verdict, file: shardFile(key), slices }; }); } @@ -2507,6 +2534,29 @@ export function runPaidReport(reportDir: string, options: { writeDurations?: boo const laterPanels = attempts.slice(1).flatMap(attempt => panelReports(manifest, artifacts.map(a => a.result), attempt) .filter(panel => panel.trials.length > 0)); + // PR-lane receipts from verdicts (the planner ships them to the next run): + // a whole fresh PASS panel with one input identity becomes a panel receipt; + // a FAIL panel or a failed rule shard becomes a negative receipt that + // blocks reuse of any older PASS for the same identity. + const revision = env.GITHUB_SHA ?? ''; + if (env.GITHUB_RUN_ID && /^[a-f0-9]{40}$/.test(revision)) { + const receiptsDir = path.join(reportDir, 'report-receipts'); + const source = { runId: `${env.GITHUB_RUN_ID}/${primary}`, revision, completedAt: Date.now() }; + const outcomes = results.flatMap(r => r.outcomes); + for (const panel of panels) { + const keys = outcomes.filter(o => o.trial?.case === panel.case && shardFile(o.files[0] ?? '') === panel.file).map(o => o.reused ? null : o.inputKey ?? null); + if (keys.length !== panel.panel.n || keys.some(k => k === null) || new Set(keys).size !== 1) continue; + if (panel.status === 'PASS') { + writePanelReceipt(receiptsDir, { schema: 1, key: keys[0]!, case: panel.case, kind: panel.kind, panel: panel.panel, source, + trials: panel.trials.map(({ trial, outcome, failure_class, exit_reason, error }) => ({ trial, outcome, + ...(failure_class ? { failure_class } : {}), ...(exit_reason ? { exit_reason } : {}), ...(error ? { error } : {}) })) }); + } else if (panel.status === 'FAIL') writeNegativeReceipt(receiptsDir, { schema: 1, key: keys[0]!, source }); + } + for (const outcome of outcomes.filter(o => !o.trial && o.inputKey && !o.reused && o.status !== 'passed')) { + writeNegativeReceipt(receiptsDir, { schema: 1, key: outcome.inputKey!, source }); + } + } + // Quarantine cap and expiry are the weekly pass-rates gate's (eval-flake-rank --gate). // History: one trial-outcomes line per isolated trial and per JUnit rule/judge case. @@ -2796,6 +2846,12 @@ async function main(): Promise { ); for (const line of formatSlicePlan(manifest)) console.log(line); for (const line of formatCapacityPreflight(manifest, options.maxParallel ?? undefined)) console.log(line); + // Planner-side reuse: ship ONE filtered receipt set with the plan, so + // every trial of a panel (on any slice) sees the same receipts. + if (manifest.profile === 'pr' && process.env.EVALS_CACHE_DIR) { + const shipped = selectPlanReceipts(process.env.EVALS_CACHE_DIR, path.join(path.dirname(path.resolve(options.emitPlanPath)), 'receipts')); + console.log(`[test:paid] reuse: shipped ${shipped.shipped} receipt(s) with the plan; blocked ${shipped.blocked.length} (newer FAIL or partial panel)`); + } return 0; } @@ -2865,6 +2921,7 @@ async function main(): Promise { const { registered, known } = fileCaseRegistration(file, fs.readFileSync(path.join(ROOT, file), 'utf8')); const exclude = mine.find(entry => entry.file === key)?.excludeCases; return prepareE2EShardReuse({ root: ROOT, key, file, caseIds: prProfileShardIds(key, manifest.selection!, exclude), + ...(trials[normalizeRelativePath(key)] ? { panel: trials[normalizeRelativePath(key)] } : {}), registeredIds: registered, registrationKnown: known, casePattern: prProfileTestNamePattern(key, manifest.selection!, exclude), expectedCases: expectedPrCaseCount(key, manifest.selection!, exclude), retries: retriesForFiles(files), timeoutMs: budget.timeoutMs, withinShardConcurrency: options.withinShardConcurrency, @@ -2901,9 +2958,9 @@ async function main(): Promise { attempt: Number.isSafeInteger(attempt) && attempt > 0 ? attempt : 1, startedAt, finishedAt: Date.now(), - outcomes: guarded.map(({ files, status, exitCode, elapsedMs, executedTests, skippedTests, budget, reused, runnerError, trial }) => + outcomes: guarded.map(({ files, status, exitCode, elapsedMs, executedTests, skippedTests, budget, reused, runnerError, trial, inputKey }) => ({ files, status, exitCode, elapsedMs, executedTests, skippedTests, ...(budget ? { budget } : {}), ...(reused ? { reused } : {}), - ...(runnerError !== undefined ? { runnerError } : {}), ...(trial ? { trial } : {}) })), + ...(runnerError !== undefined ? { runnerError } : {}), ...(trial ? { trial } : {}), ...(inputKey ? { inputKey } : {}) })), }; fs.mkdirSync(evalDirBase, { recursive: true }); const sliceResultPath = path.join(evalDirBase, `slice-${options.sliceIndex}.json`); diff --git a/scripts/typecheck-test-baseline.json b/scripts/typecheck-test-baseline.json index 612942cf7..c96077b9e 100644 --- a/scripts/typecheck-test-baseline.json +++ b/scripts/typecheck-test-baseline.json @@ -36,7 +36,6 @@ "browse/test/cookie-import-transport.test.ts\tTS2741\tProperty 'preconnect' is missing in type '() => Promise' but required in type 'typeof fetch'.": 1, "browse/test/dia-gui-readiness.test.ts\tTS2352\tConversion of type '() => { status: number; stdout: string; stderr: string; }' to type '{ (command: string): SpawnSyncReturns; (command: string, options: SpawnSyncOptionsWithStringEncoding): SpawnSyncReturns<...>; (command: string, options: SpawnSyncOptionsWithBufferEncoding): SpawnSyncReturns<...>; (command: string, options?: SpawnSyncOptions | undefined): SpawnSyncReturns<...>; (comm...' may be a mistake because neither type sufficiently overlaps with the other. If this was intentional, convert the expression to 'unknown' first. Type '{ status: number; stdout: string; stderr: string; }' is missing the following properties from type 'SpawnSyncReturns': pid, output, signal": 1, "browse/test/dia-launch-comparison.test.ts\tTS7016\tCould not find a declaration file for module '../../.github/scripts/dia-launch-driver.mjs'. '.github/scripts/dia-launch-driver.mjs' implicitly has an 'any' type.": 1, - "browse/test/dia-macos-qualification.test.ts\tTS2352\tConversion of type '() => { error?: undefined; status: number; stdout: string; stderr: string; } | { status: null; stdout: null; stderr: null; error: Error; }' to type '{ (command: string): SpawnSyncReturns; (command: string, options: SpawnSyncOptionsWithStringEncoding): SpawnSyncReturns<...>; (command: string, options: SpawnSyncOptionsWithBufferEncoding): SpawnSyncReturns<...>; (command: string, options?: SpawnSyncOptions | undefined): SpawnSyncReturns<...>; (comm...' may be a mistake because neither type sufficiently overlaps with the other. If this was intentional, convert the expression to 'unknown' first. Type '{ error?: undefined; status: number; stdout: string; stderr: string; } | { status: null; stdout: null; stderr: null; error: Error; }' is not comparable to type 'SpawnSyncReturns'. Type '{ status: null; stdout: null; stderr: null; error: Error; }' is missing the following properties from type 'SpawnSyncReturns': pid, output, signal": 2, "browse/test/dia-macos-qualification.test.ts\tTS2352\tConversion of type '() => { status: number; stdout: string; stderr: string; }' to type '{ (command: string): SpawnSyncReturns; (command: string, options: SpawnSyncOptionsWithStringEncoding): SpawnSyncReturns<...>; (command: string, options: SpawnSyncOptionsWithBufferEncoding): SpawnSyncReturns<...>; (command: string, options?: SpawnSyncOptions | undefined): SpawnSyncReturns<...>; (comm...' may be a mistake because neither type sufficiently overlaps with the other. If this was intentional, convert the expression to 'unknown' first. Type '{ status: number; stdout: string; stderr: string; }' is missing the following properties from type 'SpawnSyncReturns': pid, output, signal": 2, "browse/test/dia-macos-qualification.test.ts\tTS2769\tNo overload matches this call. The last overload gave the following error. Argument of type 'string' is not assignable to parameter of type '\"browser_profile_unavailable\" | \"code_signing_error\" | \"debugging_pipe_unavailable\" | \"default_profile_policy\" | \"dynamic_library_error\" | \"graphics_or_bootstrap_error\" | \"keychain_access_failed\" | \"keychain_interaction_disallowed\" | \"keychain_interaction_required\"'.": 1, "browse/test/domain-skills-e2e.test.ts\tTS2339\tProperty 'cleanup' does not exist on type 'BrowserManager'.": 1, diff --git a/test/ci-eval-cache.test.ts b/test/ci-eval-cache.test.ts index a34f430f5..ce2fe5618 100644 --- a/test/ci-eval-cache.test.ts +++ b/test/ci-eval-cache.test.ts @@ -21,19 +21,31 @@ test('only PR runs select the fast profile; manual and scheduled coverage stays } }); -test('receipt transport restores only this repository and PR with no broad fallback key', () => { - const steps = paid.jobs['eval-slices'].steps; - const restore = steps.filter((s: any) => s.uses?.startsWith('actions/cache/restore@')); - const save = steps.filter((s: any) => s.uses?.startsWith('actions/cache/save@')); +test('receipt transport: the planner restores only this repository and PR, the report saves one merged store', () => { + const planner = paid.jobs['plan-slices'].steps; + const restore = planner.filter((s: any) => s.uses?.startsWith('actions/cache/restore@')); expect(restore).toHaveLength(1); - expect(save).toHaveLength(1); expect(restore[0].if).toBe("github.event_name == 'pull_request'"); + expect(restore[0].with.path).toBe('/tmp/gstack-eval-input-cache'); expect(restore[0].with['restore-keys']).toBe('eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-'); - expect(save[0].with.key).toBe(restore[0].with.key); - expect(save[0].with.key).toContain('${{ github.run_id }}-${{ github.run_attempt }}-${{ matrix.slice }}'); + const emit = planner.find((s: any) => s.run?.includes('--emit-plan /tmp/paid-plan/manifest.json')); + expect(emit.env.EVALS_CACHE_DIR).toBe("${{ github.event_name == 'pull_request' && '/tmp/gstack-eval-input-cache' || '' }}"); + const upload = planner.find((s: any) => s.with?.name === 'paid-plan'); + expect(upload.with.path.trim().split('\n')).toEqual(['/tmp/paid-plan/manifest.json', '/tmp/paid-plan/receipts']); + // Executors never restore or save a cache of their own: every slice sees the plan's one receipt set. + const executor = paid.jobs['eval-slices'].steps; + expect(executor.filter((s: any) => s.uses?.startsWith('actions/cache/'))).toHaveLength(0); + expect(executor.find((s: any) => s.name === "Seed this slice's receipts from the plan").run).toContain('cp -a /tmp/paid-plan/receipts/. /tmp/paid-slice-results/receipts/'); + const report = paid.jobs['slices-report'].steps; + const merge = report.find((s: any) => s.name === "Merge this run's receipts"); + expect(merge.run).toContain('scripts/e2e-shard-reuse.ts merge /tmp/gstack-eval-input-cache'); + const save = report.filter((s: any) => s.uses?.startsWith('actions/cache/save@')); + expect(save).toHaveLength(1); expect(save[0].with.path).toBe('/tmp/gstack-eval-input-cache'); - expect(save[0].if).toContain("steps.receipts.outputs.present == 'true'"); + expect(save[0].with.key).toBe('eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-${{ github.run_id }}-${{ github.run_attempt }}-merged'); + expect(report.indexOf(save[0])).toBeGreaterThan(report.indexOf(merge)); expect(paid.jobs['eval-slices'].permissions).toEqual({ contents: 'read', packages: 'read' }); + expect(paid.jobs['slices-report'].permissions).toEqual({ contents: 'read' }); expect(JSON.stringify(periodic)).not.toContain('actions/cache/'); }); @@ -43,56 +55,21 @@ test('the judge binds cache receipts to the PR and installed runtime, not the co expect(runtime.run).toContain('sha256sum /tmp/eval-runtime-manifest.json'); const run = paid.jobs['eval-slices'].steps.find((s: any) => s.run?.includes('--plan /tmp/paid-plan/manifest.json')); expect(run.env).toMatchObject({ - EVALS_CACHE_DIR: '/tmp/gstack-eval-input-cache', + EVALS_CACHE_DIR: '/tmp/paid-slice-results/receipts', EVALS_CACHE_REPOSITORY: '${{ github.repository }}', EVALS_CACHE_PR: '${{ github.event.pull_request.number }}', EVALS_CACHE_RUNTIME_ID: '${{ needs.build-image.outputs.runtime-id }}', }); }); -test.skipIf(!Bun.which('jq') || !Bun.which('bash'))('only a new passing producer can publish the next cache snapshot', () => { - const directory = mkdtempSync(join(tmpdir(), 'ci-cache-producer-')); - const receipts = join(directory, 'receipts'); - const output = join(directory, 'output'); - mkdirSync(receipts); - const step = paid.jobs['eval-slices'].steps.find((s: any) => s.id === 'receipts'); - const script = step.run.replaceAll('/tmp/gstack-eval-input-cache', receipts); - const run = () => { - writeFileSync(output, ''); - const result = spawnSync('bash', ['-e', '-c', script], { - env: { ...process.env, GITHUB_OUTPUT: output, GITHUB_RUN_ID: '42', GITHUB_RUN_ATTEMPT: '2' }, - encoding: 'utf8', timeout: 5000, - }); - expect(result.status, result.stderr).toBe(0); - return readFileSync(output, 'utf8'); - }; - try { - expect(run()).toBe(''); - writeFileSync(join(receipts, 'old.json'), JSON.stringify({ proof: { source: { runId: '41/1' } } })); - writeFileSync(join(receipts, 'corrupt.json'), '{'); - expect(run()).toBe(''); - writeFileSync(join(receipts, 'prior-attempt.json'), JSON.stringify({ proof: { source: { runId: '42/1' } } })); - expect(run()).toBe(''); - writeFileSync(join(receipts, 'fresh.json'), JSON.stringify({ proof: { source: { runId: '42/2' } } })); - expect(run()).toBe('present=true\n'); - } finally { rmSync(directory, { recursive: true, force: true }); } -}); - -test.skipIf(!Bun.which('jq'))('the actual comment separates reused evidence, retry outcomes and deferred coverage', () => { +test.skipIf(!Bun.which('jq'))('the actual comment shows deferred coverage and never recomputes a verdict', () => { const comment = paid.jobs['slices-comment'].steps.find((s: any) => s.name === 'Post PR comment').run as string; const evaluate = (filter: string, value: unknown) => { const result = spawnSync('jq', ['-r', filter], { input: JSON.stringify(value), encoding: 'utf8', timeout: 5000 }); expect(result.status, result.stderr).toBe(0); return result.stdout.trim(); }; - const stats = comment.match(/STATS=\$\(jq -r '([^']+)'/)![1]!; - expect(evaluate(stats, { tests: [ - { name: 'retry', passed: false }, { name: 'retry', passed: true }, - { name: 'exhausted', passed: false }, { name: 'exhausted', passed: false }, - { name: 'regressed', passed: true }, { name: 'regressed', passed: false }, - { name: 'reused', passed: true, execution: 'reused' }, - ], flaky_retries: ['retry', 'exhausted', 'regressed'].map(name => ({ name, attempts: 2 })) })).toBe('4 2 2 3 3 1'); - expect(comment).toContain("printf ' | ⚠ %s cases with multiple attempts'"); + expect(comment).not.toContain('group_by(.name)'); expect(comment).not.toMatch(/flaky pass\(es\)|passed only on retry|not blocking/); const coverage = comment.match(/COVERAGE=\$\(jq -r '([^']+)'/)![1]!; const text = evaluate(coverage, { profile: 'pr', selection: { e2e: ['probe'], judges: ['judge'] }, @@ -107,9 +84,9 @@ test.skipIf(!Bun.which('jq') || !Bun.which('bash'))('comment consumes verified f const job = paid.jobs['slices-comment']; expect(job.permissions).toMatchObject({ 'pull-requests': 'write' }); expect(JSON.stringify(job.steps)).not.toMatch(/actions\/checkout|setup-bun|bun run|npm |node /); - const upload = paid.jobs['slices-report'].steps.find((step: any) => step.with?.name === 'report-verdict'); - expect(upload.with.path.trim().split('\n')).toEqual(['/tmp/report.txt', '/tmp/paid-report/collector-outcomes.json']); - expect(job.steps.find((step: any) => step.with?.name === 'report-verdict').with.path).toBe('/tmp/verdict'); + const upload = paid.jobs['slices-report'].steps.find((step: any) => step.with?.name === 'report-verdict-a${{ github.run_attempt }}'); + expect(upload.with.path.trim().split('\n')).toEqual(['/tmp/report.txt', '/tmp/paid-report/collector-outcomes.json', '/tmp/paid-report/report-summary.md']); + expect(job.steps.find((step: any) => step.with?.name === 'report-verdict-a${{ github.run_attempt }}').with.path).toBe('/tmp/verdict'); const root = mkdtempSync(join(tmpdir(), 'ci-comment-')); const paidDir = join(root, 'paid-report'); const verdictDir = join(root, 'verdict'); @@ -123,9 +100,11 @@ test.skipIf(!Bun.which('jq') || !Bun.which('bash'))('comment consumes verified f writeFileSync(join(paidDir, 'judge.json'), JSON.stringify({ total_tests: 2, tier: 'llm-judge', shard: 1, tests: [{ name: 'manual', passed: false, manual_review: { unverified: true } }, { name: 'reused', passed: true, execution: 'reused' }], flaky_retries: [] })); - const summary = { version: 1, files: [{ file: 'judge.json', tier: 'llm-judge', shard: 1, cost: 0, + const summary = { version: 2, files: [{ file: 'judge.json', tier: 'llm-judge', shard: 1, cost: 0, total: 2, passed: 1, failed: 0, manual_accepted: 1, executed: 1, reused: 1, attempts: 2, flaky: 0 }], - totals: { total: 2, passed: 1, failed: 0, manual_accepted: 1, executed: 1, reused: 1, attempts: 2, flaky: 0 } }; + totals: { total: 2, passed: 1, failed: 0, manual_accepted: 1, executed: 1, reused: 1, attempts: 2, flaky: 0 }, + verdict: { verdict: 'GREEN' }, headline: ['[test:paid] VERDICT GREEN — lane gate/pr, attempt 1'], panels: [], + failures: ['⚠ case-x behavior PASS 2/3 (✓✗✓) t2: timeout at turn 3 — @\u200bsomeone said no'] }; mkdirSync(join(verdictDir, 'paid-report')); const summaryPath = join(verdictDir, 'paid-report/collector-outcomes.json'); const script = (job.steps.find((step: any) => step.name === 'Post PR comment').run as string) @@ -142,8 +121,11 @@ test.skipIf(!Bun.which('jq') || !Bun.which('bash'))('comment consumes verified f const verified = run(); expect(verified.status, verified.stderr).toBe(0); expect(verified.stdout).toContain('⚠ MANUAL ACCEPTED (unscored)'); - expect(verified.stdout).toContain('1 automated passed / 2 final results'); - expect(verified.stdout).toContain('0 failed, 1 manual accepted'); + expect(verified.stdout).toContain('VERDICT GREEN — lane gate/pr, attempt 1'); + expect(verified.stdout).toContain('1 executed, 1 reused** rule/judge records'); + expect(verified.stdout).toContain('1 manual accepted'); + expect(verified.stdout).toContain('### Failures and split verdicts'); + expect(verified.stdout).toContain('PASS 2/3 (✓✗✓) t2: timeout at turn 3'); const unrelatedFailure = { ...summary, files: [{ ...summary.files[0], total: 3, failed: 1, executed: 2, attempts: 3 }], totals: { ...summary.totals, total: 3, failed: 1, @@ -152,16 +134,20 @@ test.skipIf(!Bun.which('jq') || !Bun.which('bash'))('comment consumes verified f const red = run(); expect(red.status, red.stderr).toBe(0); expect(red.stdout).toContain('❌ FAIL'); - expect(red.stdout).toContain('1 failed, 1 manual accepted'); + + writeFileSync(summaryPath, JSON.stringify({ ...summary, verdict: { verdict: 'RED' } })); + const redVerdict = run(); + expect(redVerdict.status, redVerdict.stderr).toBe(0); + expect(redVerdict.stdout).toContain('❌ FAIL'); writeFileSync(summaryPath, JSON.stringify({ ...summary, totals: { ...summary.totals, manual_accepted: 2 } })); const tampered = run(); expect(tampered.status, tampered.stderr).toBe(0); - expect(tampered.stdout).toContain('manual acceptance unavailable/unverified'); + expect(tampered.stdout).toContain('verified report unavailable'); expect(tampered.stdout).not.toContain('⚠ MANUAL ACCEPTED (unscored)'); rmSync(summaryPath); const absent = run(); expect(absent.status, absent.stderr).toBe(0); - expect(absent.stdout).toContain('manual acceptance unavailable/unverified'); + expect(absent.stdout).toContain('verified report unavailable'); } finally { rmSync(root, { recursive: true, force: true }); } }); diff --git a/test/e2e-shard-reuse.test.ts b/test/e2e-shard-reuse.test.ts index a07b8f8e5..acb515321 100644 --- a/test/e2e-shard-reuse.test.ts +++ b/test/e2e-shard-reuse.test.ts @@ -4,7 +4,8 @@ import * as os from 'node:os'; import * as path from 'node:path'; import { e2eReuseEnvironment, e2eReuseLaneProblem, e2eShardIdentity, e2eShardInputFiles, prepareE2EShardReuse, - type E2EShardReuseRequest, + mergeReceiptDirs, readPanelReceipt, selectPlanReceipts, writeNegativeReceipt, writePanelReceipt, + type E2EShardReuseRequest, type PanelReceipt, } from '../scripts/e2e-shard-reuse'; import { buildRunManifest, fileCaseRegistration, runPaidShard, verifySliceResults, type SliceResult } from '../scripts/test-paid-shards'; @@ -127,10 +128,13 @@ describe('E2E shard reuse through the runner', () => { test('a failed shard never publishes a receipt', async () => { let published = 0; const outcome = await runPaidShard([FILE], 1, 1, { rootDir: ROOT, logDir: scratch, env: laneEnv(), log: () => {}, - reuseFor: () => ({ lookup: () => null, publish: () => { published++; } }), + reuseFor: () => ({ inputKey: 'e'.repeat(64), unchanged: () => true, lookupPanelTrial: () => null, + lookup: () => null, publish: () => { published++; } }), commandFor: () => ({ command: process.execPath, args: ['-e', 'process.exit(1)'] }) }); expect(outcome.status).toBe('failed'); expect(published).toBe(0); + // The identity rides on the outcome so the report can store the FAIL as a negative receipt. + expect(outcome.inputKey).toBe('e'.repeat(64)); }); test('the report accepts reused results only in the fast PR profile', () => { @@ -143,3 +147,73 @@ describe('E2E shard reuse through the runner', () => { expect(verifySliceResults(manifest, results).problems).toContain(`${FILE}: only the fast PR profile may reuse results; this lane executes fresh`); }); }); + +describe('planner-side panel reuse and negative receipts', () => { + const panelPlan = { kind: 'behavior' as const, panel: { n: 3, k: 2 }, quarantined: false }; + const trialRequest = (trial: number, over: Partial = {}) => request({ + key: `${FILE}#setup-deploy-workflow~t${trial}`, panel: panelPlan, ...over }); + const source = (completedAt: number, runId = '1001/1') => ({ runId, revision: 'd'.repeat(40), completedAt }); + const panel = (key: string, outcomes: Array<'passed' | 'failed'>, completedAt = Date.now() - 1_000): PanelReceipt => ({ + schema: 1, key, case: 'setup-deploy-workflow', kind: 'behavior', panel: { n: 3, k: 2 }, source: source(completedAt), + trials: outcomes.map((outcome, i) => ({ trial: i + 1, outcome, ...(outcome === 'failed' ? { failure_class: 'timeout' as const } : {}) })), + }); + + test('every trial of a panel shares one identity; the panel policy is part of it', () => { + const key = (r: E2EShardReuseRequest) => { const x = e2eShardIdentity(r); if (x.status !== 'eligible') throw new Error(x.reason); return x.identity.key; }; + const t1 = key(trialRequest(1)); + expect(key(trialRequest(2, { env: laneEnv({ GSTACK_EVAL_TRIAL: '2' }) }))).toBe(t1); + expect(key(trialRequest(1, { panel: { ...panelPlan, quarantined: true } }))).not.toBe(t1); + expect(key(request())).not.toBe(t1); + }); + + test('a whole PASS panel receipt is reused per trial, a split PASS keeps its failed trial', () => { + const dir = path.join(scratch, 'panel-hit'); + const env = laneEnv({ EVALS_CACHE_DIR: dir }); + const reuse = prepareE2EShardReuse(trialRequest(2, { env }))!; + expect(reuse.lookupPanelTrial(2)).toBeNull(); + writePanelReceipt(dir, panel(reuse.inputKey, ['passed', 'failed', 'passed'])); + expect(reuse.lookupPanelTrial(2)).toMatchObject({ trial: { trial: 2, outcome: 'failed', failure_class: 'timeout' }, hit: { source: { runId: '1001/1' } } }); + expect(reuse.lookupPanelTrial(1)!.trial.outcome).toBe('passed'); + }); + + test('FAIL, partial, expired or negatively receipted panels are never reused', () => { + const dir = path.join(scratch, 'panel-miss'); + const key = 'a'.repeat(64); + for (const receipt of [panel(key, ['passed', 'failed', 'failed']), panel(key, ['passed', 'passed']), + panel(key, ['passed', 'passed', 'passed'], Date.now() - 2 * 24 * 60 * 60 * 1000)]) { + writePanelReceipt(dir, receipt); + expect(readPanelReceipt(dir, key)).toBeNull(); + } + writePanelReceipt(dir, panel(key, ['passed', 'passed', 'passed'], Date.now() - 5_000)); + expect(readPanelReceipt(dir, key)).not.toBeNull(); + writeNegativeReceipt(dir, { schema: 1, key, source: source(Date.now() - 1_000, '1002/1') }); + expect(readPanelReceipt(dir, key)).toBeNull(); + }); + + test('the planner ships one filtered set: a newer FAIL blocks an older PASS, an older FAIL does not', () => { + const from = path.join(scratch, 'select-from'); + const to = path.join(scratch, 'select-to'); + fs.mkdirSync(from, { recursive: true }); + const [blockedKey, keptKey, panelKey] = ['1', '2', '3'].map(c => c.repeat(64)); + const passReceipt = (key: string, completedAt: number) => fs.writeFileSync(path.join(from, `${key}.json`), + JSON.stringify({ schema: 1, proof: { source: source(completedAt) } })); + passReceipt(blockedKey, 1_000); + writeNegativeReceipt(from, { schema: 1, key: blockedKey, source: source(2_000, '1002/1') }); + passReceipt(keptKey, 3_000); + writeNegativeReceipt(from, { schema: 1, key: keptKey, source: source(2_000, '1002/1') }); + writePanelReceipt(from, panel(panelKey, ['passed', 'passed'])); + const result = selectPlanReceipts(from, to); + expect(result.blocked.sort()).toEqual([`${blockedKey}.json`, `${panelKey}.panel.json`].sort()); + expect(fs.readdirSync(to).sort()).toEqual([`${blockedKey}.fail.json`, `${keptKey}.fail.json`, `${keptKey}.json`].sort()); + }); + + test('merging receipt stores keeps the newest file per name', () => { + const [a, b, out] = ['merge-a', 'merge-b', 'merge-out'].map(name => path.join(scratch, name)); + const key = '4'.repeat(64); + writeNegativeReceipt(a, { schema: 1, key, source: source(5_000, '1/1') }); + writeNegativeReceipt(b, { schema: 1, key, source: source(9_000, '2/1') }); + expect(mergeReceiptDirs(out, [a, b, path.join(scratch, 'missing')])).toBe(2); + expect(JSON.parse(fs.readFileSync(path.join(out, `${key}.fail.json`), 'utf8')).source.runId).toBe('2/1'); + expect(mergeReceiptDirs(out, [a])).toBe(0); + }); +}); diff --git a/test/paid-report-fail-open.test.ts b/test/paid-report-fail-open.test.ts index 27343310c..86b9cc61e 100644 --- a/test/paid-report-fail-open.test.ts +++ b/test/paid-report-fail-open.test.ts @@ -31,7 +31,7 @@ function manifest(entries: PaidRunManifest['entries'], sliceCount: number): Paid } let caseCounter = 0; -function report(plan: PaidRunManifest, slices: SliceResult[], collectors: Record = {}) { +function report(plan: PaidRunManifest, slices: SliceResult[], collectors: Record = {}, env: NodeJS.ProcessEnv = {}) { const dir = path.join(base, `case-${++caseCounter}`); fs.mkdirSync(dir, { recursive: true }); fs.writeFileSync(path.join(dir, 'manifest.json'), JSON.stringify(plan)); @@ -41,7 +41,7 @@ function report(plan: PaidRunManifest, slices: SliceResult[], collectors: Record fs.writeFileSync(path.join(dir, name), JSON.stringify(body)); } const result = spawnSync(process.execPath, [path.join(ROOT, 'scripts/test-paid-shards.ts'), '--tier', plan.tier, '--report', dir], - { cwd: ROOT, encoding: 'utf8', timeout: 30_000, env: { ...process.env, EVALS_TIER: plan.tier } }); + { cwd: ROOT, encoding: 'utf8', timeout: 30_000, env: { ...process.env, GITHUB_RUN_ID: '', GITHUB_SHA: '', EVALS_TIER: plan.tier, ...env } }); return { status: result.status, out: `${result.stdout}\n${result.stderr}`, dir }; } @@ -219,4 +219,19 @@ describe('behavior and quarantined panels through --report', () => { expect(again.status).toBe(1); expect(again.stdout).toContain('attempt 1 (later attempts 2 reported, never replacing it)'); }); + + test('verdicts become next-run receipts: a whole fresh PASS panel, and negatives for FAIL panels', () => { + const inputKey = 'b'.repeat(64); + const withKey = (results: TrialResult[]) => [1, 2, 3, 4].map(index => slice(index, 4, index === 4 ? [passed(RULE_A)] + : [{ ...trialOutcome(index, results[index - 1]!)!, inputKey }])); + const env = { GITHUB_RUN_ID: '77', GITHUB_SHA: 'e'.repeat(40) }; + const green = report(plan(), withKey(['passed', 'failed', 'passed']), {}, env); + expect(green.status, green.out).toBe(0); + const receipt = JSON.parse(fs.readFileSync(path.join(green.dir, 'report-receipts', `${inputKey}.panel.json`), 'utf8')); + expect(receipt).toMatchObject({ key: inputKey, case: ID, panel: { n: 3, k: 2 }, source: { runId: '77/1' } }); + expect(receipt.trials.map((t: any) => t.outcome)).toEqual(['passed', 'failed', 'passed']); + const red = report(plan(), withKey(['passed', 'failed', 'failed']), {}, env); + expect(red.status).toBe(1); + expect(fs.readdirSync(path.join(red.dir, 'report-receipts'))).toEqual([`${inputKey}.fail.json`]); + }); }); diff --git a/test/paid-run-manifest.test.ts b/test/paid-run-manifest.test.ts index 74090aace..ddd9b96ce 100644 --- a/test/paid-run-manifest.test.ts +++ b/test/paid-run-manifest.test.ts @@ -39,6 +39,7 @@ import { verifySliceResults, expandTrialShards, formatCapacityPreflight, + panelReports, shardSlug, type PaidRunManifest, type ShardOutcome, @@ -607,4 +608,17 @@ describe('trial planner (behavior and quarantined panels)', () => { expect(packed.slices).toHaveLength(3); expect(Object.values(packed.estimates)).toEqual([180_000, 180_000, 180_000]); }); + + test('reuse is whole-panel only: a panel mixing reused and fresh trials is INCOMPLETE', () => { + const manifest = budgetPlan('gate', { 'review-sql-injection': 'behavior' }); + const trials = manifest.entries.filter(entry => entry.trial); + const reused = { inputKey: 'c'.repeat(64), runId: '1001/1', revision: 'd'.repeat(40), completedAt: 1 }; + const results = (reusedTrials: number[]): SliceResult[] => trials.map(entry => ({ version: 1, tier: 'gate', sliceIndex: entry.slice, + sliceCount: manifest.sliceCount, outcomes: [{ files: [entry.file], status: 'passed', exitCode: 0, elapsedMs: 1, executedTests: 1, skippedTests: 0, + trial: { case: 'review-sql-injection', trial: Number(entry.file.slice(-1)), ...entry.trial!, outcome: 'passed', cost_usd: 0, duration_ms: 1 }, + ...(reusedTrials.includes(Number(entry.file.slice(-1))) ? { reused } : {}) }] })); + expect(panelReports(manifest, results([]), 1)[0]).toMatchObject({ status: 'PASS' }); + expect(panelReports(manifest, results([1, 2, 3]), 1)[0]).toMatchObject({ status: 'PASS' }); + expect(panelReports(manifest, results([2]), 1)[0]).toMatchObject({ status: 'INCOMPLETE', failsLane: true, reason: expect.stringContaining('partial panel reuse') }); + }); }); From 32772f535d683e488d774f8b045135d6cd221a0f Mon Sep 17 00:00:00 2001 From: garrytan Date: Tue, 29 Sep 2026 19:44:55 +0000 Subject: [PATCH 15/15] feat(evals): --case/--trials local diagnosis and panels in local sharded runs bun run scripts/test-paid-shards.ts --case [--trials N] runs N independent trials of one case through the CI panel runner (trial shards, TRIAL_ENV identity, name-pattern isolation) and prints its panelVerdict(); N defaults to the case's policy panel and CI never reads it. The local sharded path (test:gate:sharded, test:periodic:sharded) now plans the same trial shards and exclusions as CI and exits on execution completeness plus panel verdicts. --- scripts/test-paid-shards.ts | 117 ++++++++++++++++++++++++++++++++---- test/paid-shards.test.ts | 25 ++++++++ 2 files changed, 131 insertions(+), 11 deletions(-) diff --git a/scripts/test-paid-shards.ts b/scripts/test-paid-shards.ts index 1a5d9c8b4..cf853b04f 100644 --- a/scripts/test-paid-shards.ts +++ b/scripts/test-paid-shards.ts @@ -714,14 +714,15 @@ export function partitionShardsByDiffSelection( export function planPaidShards( files: string[], - options: { maxFilesPerShard?: number } = {}, + options: { maxFilesPerShard?: number; ownShard?: ReadonlySet } = {}, ): string[][] { const size = Math.max(1, options.maxFilesPerShard ?? DEFAULT_MAX_FILES_PER_SHARD); const unique = [...new Set(files.map(normalizeRelativePath))].sort(); const shards: string[][] = []; let pending: string[] = []; for (const file of unique) { - if (isOverlayTestFile(file) || shardCaseId(file) !== null || FILE_RETRY_BUDGETS.some(budget => budget.file === shardFile(file))) { + if (isOverlayTestFile(file) || shardCaseId(file) !== null || FILE_RETRY_BUDGETS.some(budget => budget.file === shardFile(file)) + || options.ownShard?.has(file)) { if (pending.length) shards.push(pending); pending = []; shards.push([file]); @@ -2241,6 +2242,55 @@ export function collectorOutcomeCounts(results: Array<{ tests?: Array<{ } +// ─── Local diagnosis: one case through the CI panel runner (A9) ──────────── + +/** The one paid file that statically registers `id`, else a thrown reason. */ +export function caseFile(id: string, rootDir = ROOT, discovered = collectPaidTestFiles(rootDir)): string { + const owners = discovered.filter(file => { + const { registered, known } = fileCaseRegistration(file, fs.readFileSync(path.join(rootDir, file), 'utf8')); + return known && registered.includes(id); + }); + if (owners.length !== 1) throw new Error(`--case ${id}: ${owners.length ? `registered by ${owners.join(', ')}` : 'no paid file statically registers it'}; it needs exactly one`); + return owners[0]!; +} + +/** + * Run `trials` independent trials of one case exactly as CI runs a panel + * (trial shards, TRIAL_ENV identity, name-pattern isolation), then print its + * panelVerdict(). `trials` defaults to the case's policy panel; CI never + * reads it. k keeps the kind's meaning (all trials for rule, the policy + * majority for behavior, capped at n). + */ +export async function runCaseDiagnosis(id: string, options: { + trials?: number; jobs?: number; withinShardConcurrency?: number; timeoutMs?: number; rootDir?: string; env?: NodeJS.ProcessEnv; + evalDirBase?: string; commandFor?: RunShardsOptions['commandFor']; log?: (line: string) => void; file?: string; +} = {}): Promise { + const rootDir = options.rootDir ?? ROOT; + const log = options.log ?? ((line: string) => console.log(line)); + const file = options.file ?? caseFile(id, rootDir); + const policy = caseTrialPlan(id); + const n = options.trials ?? policy.panel.n; + const plan: CaseTrialPlan = { ...policy, panel: { n, k: policy.panel.k === policy.panel.n ? n : Math.min(policy.panel.k, n) } }; + const keys = Array.from({ length: n }, (_, i) => trialShardKey(file, id, i + 1)); + const tier = E2E_TIERS[id] as PaidTier; + log(`[test:paid] --case ${id}: ${n} trial(s) of ${file} (kind ${plan.kind}, PASS at ${plan.panel.k}/${n}${plan.quarantined ? ', quarantined' : ''}), tier=${tier}`); + const summary = await runPaidShards(keys.map(key => [key]), { + jobs: Math.min(options.jobs ?? DEFAULT_JOBS, n), withinShardConcurrency: options.withinShardConcurrency, timeoutMs: options.timeoutMs, + rootDir, log, commandFor: options.commandFor, evalDirBase: options.evalDirBase, + trials: Object.fromEntries(keys.map(key => [key, plan])), + env: { ...(options.env ?? process.env), EVALS: '1', EVALS_TIER: tier, EVALS_PREFLIGHT_OK: '1', EVALS_ALL: '1', + ...paidSelectionEnv('full', { e2e: [id], judges: [] }, `--case ${id}`) }, + }); + const trials = summary.outcomes.flatMap(outcome => outcome.trial && outcome.trial.outcome !== null ? [{ + trial: outcome.trial.trial, outcome: outcome.trial.outcome, ...(outcome.trial.failure_class ? { failure_class: outcome.trial.failure_class } : {}), + ...(outcome.trial.exit_reason ? { exit_reason: outcome.trial.exit_reason } : {}), ...(outcome.trial.error ? { error: outcome.trial.error } : {}), + ...(outcome.trial.timeout_at_turn !== undefined ? { timeout_at_turn: outcome.trial.timeout_at_turn } : {}) }] : []); + const verdict = panelVerdict({ case: id, kind: plan.kind, panel: plan.panel, trials, quarantined: plan.quarantined }); + log(formatPanelLine({ ...verdict, file, slices: {} }, tier)); + for (const outcome of summary.outcomes.filter(o => o.trial?.outcome === null)) log(` t${outcome.trial!.trial}: no trial record (${outcome.trial!.harness})`); + return verdict; +} + // ─── Report: verdicts, history records and the human readout ─────────────── /** One Bun JUnit testcase (`--reporter=junit`). */ @@ -2716,6 +2766,9 @@ type CliOptions = { writeDurations: boolean; /** Planner: the workflow matrix cap, for the capacity preflight's wave count. */ maxParallel: number | null; + /** Local diagnosis: run one case through the panel runner CI uses (never read by CI). */ + caseId: string | null; + trials: number | null; }; function parsePositiveInt(value: string | undefined, flag: string): number { @@ -2771,6 +2824,8 @@ export function parseCliOptions(argv: string[], env: NodeJS.ProcessEnv = process reportDir: null, writeDurations: false, maxParallel: null, + caseId: null, + trials: null, }; for (let index = 0; index < argv.length; index += 1) { @@ -2809,9 +2864,19 @@ export function parseCliOptions(argv: string[], env: NodeJS.ProcessEnv = process } if (arg === '--write-durations') { options.writeDurations = true; continue; } if (arg === '--max-parallel') { options.maxParallel = parsePositiveInt(argv[index += 1], '--max-parallel'); continue; } + if (arg === '--case') { + const value = argv[index += 1]; + if (!value || !Object.hasOwn(E2E_TIERS, value)) throw new Error(`--case needs a live E2E case id. Received: ${value}`); + options.caseId = value; continue; + } + if (arg === '--trials') { options.trials = parsePositiveInt(argv[index += 1], '--trials'); continue; } throw new Error(`Unknown argument: ${arg}`); } if (options.writeDurations && !options.reportDir) throw new Error('--write-durations requires --report'); + if (options.trials !== null && options.caseId === null) throw new Error('--trials requires --case'); + if (options.caseId !== null && (options.emitPlanPath || options.planPath || options.reportDir || options.sliceIndex !== null)) { + throw new Error('--case is local diagnosis; it cannot combine with --emit-plan, --plan/--slice or --report'); + } if (options.sliceBudgetMs !== null && argv.includes('--slices')) throw new Error('Plan with exactly one of --slices or --slice-budget'); if (options.sliceBudgetMs !== null && !options.jobsExplicit) throw new Error('--slice-budget needs explicit --jobs (or EVALS_JOBS): the plan packs and supervises for that worker count'); if (options.skipJudges && (!options.emitPlanPath || options.tier !== 'gate')) throw new Error('--skip-judges applies only to an emitted gate census plan'); @@ -2859,6 +2924,14 @@ async function main(): Promise { // a slice whose artifact never landed is a FAILURE, not an absence. if (options.reportDir) return runPaidReport(options.reportDir, { writeDurations: options.writeDurations }); + if (options.caseId) { + preflightAnthropicApi(process.env); + const verdict = await runCaseDiagnosis(options.caseId, { trials: options.trials ?? undefined, jobs: options.jobs, + withinShardConcurrency: options.withinShardConcurrency, timeoutMs: timeoutOverride, + evalDirBase: process.env.GSTACK_EVAL_DIR || getProjectEvalDir() }); + return verdict.status === 'PASS' ? 0 : 1; + } + const discovered = collectPaidTestFiles(); if (discovered.length === 0) throw new Error('No paid test files were discovered.'); @@ -2986,18 +3059,22 @@ async function main(): Promise { const caseKeys = partitionCaseExclusions(expandCaseShards(tierSelection.selected, options.tier)); const selected = tierSelection.selected; const excluded = [...tierSelection.excluded, ...caseKeys.excluded]; - const shards = planPaidShards(caseKeys.runnable, { maxFilesPerShard: options.maxFilesPerShard }); + // Same panels as CI: isolated cases run as trial shards, their file shard excludes them. + const expansion = expandTrialShards(caseKeys.runnable, options.tier); + const shards = planPaidShards(expansion.keys, { maxFilesPerShard: options.maxFilesPerShard, + ownShard: new Set(Object.keys(expansion.excludeCases)) }); // Parent-side diff selection (D9): skip whole shards whose mapped tests are // all unselected. Fail-open everywhere — the child's self-skip stays // authoritative for anything the mapper can't attribute. const cases = computePaidCaseSelection({ profile: options.profile }); const fast = cases.coverage?.mode === 'pr'; - const profileShards = fast ? shards.filter(files => files.some(file => prProfileFileSelected(file, cases.selection))) : shards; + const excludeOf = (file: string) => expansion.excludeCases[normalizeRelativePath(file)] ?? []; + const profileShards = fast ? shards.filter(files => files.some(file => prProfileFileSelected(file, cases.selection, excludeOf(file)))) : shards; const { runnable, skipped } = partitionShardsByDiffSelection(profileShards, - cases.selection.e2e === null ? null : new Set(cases.selection.e2e)); + cases.selection.e2e === null ? null : new Set(cases.selection.e2e), { excludeCases: expansion.excludeCases }); if (fast) for (const files of shards) { - if (!files.some(file => prProfileFileSelected(file, cases.selection))) skipped.push({ files, reason: 'Outside the fast PR profile; retained in broad coverage' }); + if (!files.some(file => prProfileFileSelected(file, cases.selection, excludeOf(file)))) skipped.push({ files, reason: 'Outside the fast PR profile; retained in broad coverage' }); } const selectedCount = cases.selection.e2e?.length ?? Object.keys(E2E_TOUCHFILES).length; console.log( @@ -3038,9 +3115,11 @@ async function main(): Promise { timeoutMs: options.timeoutExplicit ? options.timeoutMs : undefined, jobs: options.jobs, withinShardConcurrency: options.withinShardConcurrency, + trials: expansion.trials, + casePatterns: Object.fromEntries(Object.entries(expansion.excludeCases).map(([file, ids]) => [file, excludedCasesNamePattern(ids)])), ...(fast ? { - expectedCases: Object.fromEntries(runnable.flat().map(file => [file, expectedPrCaseCount(file, cases.selection)])), - casePatterns: Object.fromEntries(runnable.flat().map(file => [file, prProfileTestNamePattern(file, cases.selection)])), + expectedCases: Object.fromEntries(runnable.flat().map(file => [file, expectedPrCaseCount(file, cases.selection, excludeOf(file))])), + casePatterns: Object.fromEntries(runnable.flat().map(file => [file, prProfileTestNamePattern(file, cases.selection, excludeOf(file))])), } : {}), env: { ...process.env, @@ -3065,13 +3144,29 @@ async function main(): Promise { executedTests: null, skippedTests: null, })); - const guardedOutcomes = applyHollowShardGuard(runSummary.outcomes, { + const guardedOutcomes = guardTrialRecords(applyHollowShardGuard(runSummary.outcomes, { evalsAll: process.env.EVALS_ALL === '1', requireExecuted: fast, - }); + })); const summary = summarize([...guardedOutcomes, ...skippedOutcomes]); for (const line of formatSummary(summary)) console.log(line); - return summaryExitCode(summary); + // The same verdict rule as the CI report: one panelVerdict() per isolated case. + const panels = new Map(); + for (const outcome of guardedOutcomes) { + const panel = outcome.trial ? trialPanelKey(outcome.files[0]!) : null; + if (panel) panels.set(panel, [...(panels.get(panel) ?? []), outcome.trial!]); + } + let panelRed = false; + for (const [key, records] of panels) { + const plan = expansion.trials[`${key}~t1`]!; + const verdict = panelVerdict({ case: shardCaseId(key)!, kind: plan.kind, panel: plan.panel, quarantined: plan.quarantined, + trials: records.filter(r => r.outcome !== null).map(r => ({ trial: r.trial, outcome: r.outcome!, + ...(r.failure_class ? { failure_class: r.failure_class } : {}), ...(r.exit_reason ? { exit_reason: r.exit_reason } : {}), + ...(r.error ? { error: r.error } : {}) })) }); + if (verdict.status !== 'PASS' || verdict.split) console.log(` ${formatPanelLine({ ...verdict, file: shardFile(key), slices: {} }, options.tier)}`); + panelRed ||= verdict.failsLane; + } + return sliceExitCode([...guardedOutcomes, ...skippedOutcomes]) || (panelRed ? 1 : 0); } if (import.meta.main) { diff --git a/test/paid-shards.test.ts b/test/paid-shards.test.ts index 5268eef0d..1f98f1e29 100644 --- a/test/paid-shards.test.ts +++ b/test/paid-shards.test.ts @@ -54,6 +54,9 @@ import { caseIdForTestName, shardTrial, excludedCasesNamePattern, + runCaseDiagnosis, + caseFile, + parseCliOptions, type CaseTrialPlan, type ShardOutcome, } from '../scripts/test-paid-shards'; @@ -674,4 +677,26 @@ console.log("Ran 1 tests across 1 files. [1ms]"); process.exit(${fail ? 1 : 0}); expect(caseIdForTestName(CASE_TEST_NAMES['plan-review-report']!)).toBe('plan-review-report'); expect(caseIdForTestName('plain helper')).toBeNull(); }); + + test('--case/--trials: local diagnosis flags are validated and never combine with CI modes', () => { + expect(parseCliOptions(['--case', 'review-sql-injection', '--trials', '5'], {})).toMatchObject({ caseId: 'review-sql-injection', trials: 5 }); + expect(() => parseCliOptions(['--trials', '3'], {})).toThrow('--trials requires --case'); + expect(() => parseCliOptions(['--case', 'no-such-case'], {})).toThrow('live E2E case id'); + expect(() => parseCliOptions(['--case', 'review-sql-injection', '--report', '/tmp/r'], {})).toThrow('local diagnosis'); + expect(caseFile('review-sql-injection')).toBe('test/skill-e2e-review.test.ts'); + }); + + test('--case runs the CI panel runner and prints its panelVerdict', async () => { + const evalDirBase = fs.mkdtempSync(path.join(os.tmpdir(), 'case-diagnosis-')); + const lines: string[] = []; + try { + const verdict = await runCaseDiagnosis('review-sql-injection', { trials: 3, evalDirBase, log: line => lines.push(line), jobs: 3, + commandFor: files => ({ command: process.execPath, args: ['-e', + `console.log("Ran 1 tests across 1 files. [1ms]"); process.exit(${files[0]!.endsWith('~t3') ? 1 : 0});`] }) }); + // A rule case keeps its meaning locally: every trial must pass. + expect(verdict).toMatchObject({ case: 'review-sql-injection', kind: 'rule', panel: { n: 3, k: 3 }, passed: 2, status: 'FAIL' }); + expect(lines.join('\n')).toContain('--case review-sql-injection: 3 trial(s) of test/skill-e2e-review.test.ts'); + expect(lines.join('\n')).toContain('FAIL 2/3 (✓✓✗)'); + } finally { fs.rmSync(evalDirBase, { recursive: true, force: true }); } + }); });