Files
gstack/test/helpers/eval-budgets.ts
T
Garry Tan a84b0b5b6d v1.90.0.0 feat: make browser cookie imports explicit and safe (#2964)
* fix(browse): prepare reliable cookie import wave for validation

* ci: sequence quality and behavior for validation branch

* fix(browse): isolate Windows qualification and preserve native diagnostics

* test(browse): cover cookie workflow quality and isolate Windows user paths

* test(browse): trace native member startup and initialize fresh folders

* fix(browse): keep Windows member stdin alive through EOF

* fix(browse): latch native timeouts and compare contained Edge startup

* test(browse): verify native version metadata and actual Windows argv

* test(browse): qualify Dia import on isolated macOS CI

* fix(browse): require picker origin for session mutations

* fix(browse): bound credential reads through stream completion

* test(browse): inspect owned Windows process arguments natively

* test(evals): preserve passing coverage during cookie repair reruns

* test(browse): isolate Dia qualification in a fresh macOS account

* test(browse): pass bounded integer timeouts to native Mac probes

* test(browse): distinguish Windows profile initialization from containment

* test(browse): await descendant pipe readiness before parent exit

* test(browse): initialize and restore isolated macOS Keychain state

* test(browse): initialize Windows fixture folders before qualification

* test(ci): pin the same Node runtime across Windows checks

* test(browse): distinguish native macOS browser preflight stages

* test(browse): isolate Windows descendant console lifetime

* test(browse): preserve native receipts and identify fixture lock holders

* test(browse): prepare dependency resolution before native Mac worker startup

* test(ci): include lock and close checks in native diagnostics

* test(browse): preserve native owner probe stages and subprocess deadlines

* fix(browse): classify Chromium profile-in-use exit precisely

* test(browse): retain Mac qualification evidence through cleanup failures

* test(browse): bound Mac fixture paths and retire its owned user domain

* test(browse): accept vanished fixture entries without weakening cleanup

* test(browse): identify probe-created macOS user domains safely

* test(browse): observe Mac user domains without targeting them first

* test(browse): use passive fresh-user ownership throughout Mac qualification

* test(browse): distinguish profile and registered-home Keychain lookups

* test(browse): qualify Dia under one registered account home

* test(browse): identify Dia startup and owned process-group failures

* test(browse): classify bounded Dia startup diagnostics without leaking output

* fix(test): preserve native Mac sandboxing and reap owned browser children

* fix(browse): preserve Chromium sandboxing for native profile imports

* test(browse): inspect signed Mach-O architecture without launching Xcode tools

* test(browse): sample pending Dia startup and reap on all cleanup paths

* test(browse): compare protected Dia launches in fresh Bun and Node accounts

* test(browse): inspect isolated Mac GUI readiness without browser access

* v1.90.0.0 fix: bind cookie picker actions to their document

* test: validate cookie guards and fit nested launch fixtures

* ci: configure the bundled Chromium sandbox helper

* fix(browse): classify Playwright authentication timeouts

* test: retain bounded Windows lifecycle diagnostics

* test(cso): reuse bounded NTFS precision candidates

* test(review): handle explicit preservation choices safely

* test(browse): remove owned fixture directories with explicit primitives

* test(review): distinguish descriptive reuse from edit commitments

* test: admit only the approved unscored cookie workflow refusal

* test: keep the Office Hours judge mock export-complete

* fix: keep dependency-free CI planners independent of the model SDK

* test: observe the exact holder after a native fixture unlink failure

* fix: start seeded PTY observations at owned readiness

* test: acquire identity-bound Windows deletion admission before profile resets

* test: preserve qualified Git index bits without authorizing mutations
2026-09-25 12:06:45 -04:00

141 lines
6.8 KiB
TypeScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
/**
* Timeout policy for paid tests — five tiers instead of hand-tuned sprawl.
*
* Before this module the paid suite carried 46×300s, 46×120s, 44×360s,
* 44×180s, 27×240s, 19×150s, 13×420s, 12×600s, 7×700s… hand-ratcheted
* per test, several inflated to paper over the old 40-way in-shard
* concurrency (session startup queued behind 39 siblings and ate the
* budget before turn one — dead with the sharded runner's 1-file-per-shard
* model). Pick the tier that matches the test's SHAPE; escape-hatch raw
* literals stay legal with a justification comment (count-ratcheted by
* test/eval-budgets-policy.test.ts).
*
* Every tier must fit inside the lane walls — pinned by the fit test in
* test/eval-budgets-policy.test.ts against the sharded runner's
* DEFAULT_SHARD_TIMEOUT_MS. Budget above the wall is fiction, not headroom.
*/
/** LLM-judge call over an existing capture (no agent session). */
export const JUDGE_MS = 120_000;
/** One bounded capture: SDK execution or the first displayed native question. */
export const CAPTURE_MS = 300_000;
/** Multi-capture or long multi-turn `claude -p` flows. */
export const CAPTURE_LONG_MS = 600_000;
/** Interactive real-PTY flow (spawn + skill + a few interactions). */
export const PTY_MS = 900_000;
/**
* Chained/judged PTY observation — the ceiling tier. 1200s leaves the
* 1800s shard wall real overhead; anything that genuinely needs more
* should be split or use an explicitly registered workflow exception with
* corresponding runner and CI walls; never inflate an ordinary tier.
*/
export const PTY_LONG_MS = 1_200_000;
export const ALL_TIERS = {
JUDGE_MS,
CAPTURE_MS,
CAPTURE_LONG_MS,
PTY_MS,
PTY_LONG_MS,
} as const;
/**
* Explicit exception for one uninterrupted four-phase workflow. These are
* specified allowances, not measured latency or a conservative confidence bound.
* The historical 900-second failure remains a failure. Ordinary tiers do not grow.
*/
export const AUTOPLAN_CHAIN_BUDGET = {
id: 'autoplan-four-native-phases-v1',
file: 'test/skill-e2e-autoplan-chain.test.ts',
workMs: 4 * PTY_LONG_MS,
sessionMs: 84 * 60_000,
testMs: 85 * 60_000,
shardMs: 172 * 60_000,
retries: 1,
shardReserveMs: 2 * 60_000,
ciJobMs: 200 * 60_000,
ciReserveMs: 28 * 60_000,
reason: 'One command must complete CEO, Design, DX and Eng, including native reviews and amendment handoffs.',
} as const;
/** Whole-file supervision must cover each existing attempt and its retry.
* These six fixtures already allow 25 minutes per case; the old 30-minute
* wall could kill a second attempt after five minutes. No case budget grows.
* Reserve the sequential upper bound even when Bun runs sibling cases together.
*/
export const FINDING_RETRY_BUDGETS = [
{ file: 'test/skill-e2e-plan-ceo-finding-count.test.ts', cases: 2 },
{ file: 'test/skill-e2e-plan-ceo-split-overflow.test.ts', cases: 1 },
{ file: 'test/skill-e2e-plan-design-finding-count.test.ts', cases: 1 },
{ file: 'test/skill-e2e-plan-devex-finding-count.test.ts', cases: 1 },
{ file: 'test/skill-e2e-plan-eng-finding-count.test.ts', cases: 1 },
{ file: 'test/skill-e2e-plan-eng-multi-finding-batching.test.ts', cases: 1 },
].map(({ file, cases }) => ({
file, cases,
id: `${file.slice('test/skill-e2e-'.length, -'.test.ts'.length)}-existing-retry-v1`,
testMs: 1_500_000,
retries: 1,
shardReserveMs: AUTOPLAN_CHAIN_BUDGET.shardReserveMs,
shardMs: cases * 1_500_000 * 2 + AUTOPLAN_CHAIN_BUDGET.shardReserveMs,
}));
/** Three existing captures and one configured retry; only supervision grows. */
export const AUQ_CONSISTENCY_RETRY_BUDGET = {
file: 'test/skill-e2e-auq-consistency.test.ts',
id: 'auq-consistency-existing-retry-v1',
cases: 1,
testMs: 3 * CAPTURE_MS + 60_000,
retries: 1,
shardReserveMs: AUTOPLAN_CHAIN_BUDGET.shardReserveMs,
shardMs: (3 * CAPTURE_MS + 60_000) * 2 + AUTOPLAN_CHAIN_BUDGET.shardReserveMs,
} as const;
/** These fixtures have a fixed case count in every supported tier. */
export const STRICT_RETRY_CASE_BUDGETS = [...FINDING_RETRY_BUDGETS, AUQ_CONSISTENCY_RETRY_BUDGET];
/** Whole-file walls cover all existing cases and retries, even if Bun runs them
* sequentially. Mixed-tier files reserve their larger complete tier, never a
* currently selected subset. These rows add no case-count or model-work policy.
* The 10-second terms preserve the existing Codex/recording finalization grace.
*/
export const FILE_RETRY_BUDGETS = [
...STRICT_RETRY_CASE_BUDGETS,
...[
// Fourteen workflow judges include their 10s recording grace; the other
// eleven judges retain 120s. Supervise all 25 and the existing one retry.
{ file: 'test/skill-llm-eval.test.ts', attemptMs: 15 * (JUDGE_MS + 10_000) + 11 * JUDGE_MS, retries: 1 },
{ file: 'test/codex-e2e-plan-format.test.ts', attemptMs: 4 * (CAPTURE_LONG_MS + 10_000), retries: 1 },
{ file: 'test/skill-e2e-auq-matrix.test.ts', attemptMs: 6 * CAPTURE_MS, retries: 1 },
{ file: 'test/skill-e2e-plan-format.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), retries: 1 },
{ file: 'test/skill-e2e-auto-decide-preserved.test.ts', attemptMs: PTY_MS, retries: 1 },
{ file: 'test/skill-e2e-plan-ceo-finding-floor.test.ts', attemptMs: PTY_MS, retries: 1 },
{ file: 'test/skill-e2e-plan-eng-finding-floor.test.ts', attemptMs: PTY_MS, retries: 1 },
{ file: 'test/skill-e2e-plan-design-finding-floor.test.ts', attemptMs: PTY_MS, retries: 1 },
{ file: 'test/skill-e2e-plan-devex-finding-floor.test.ts', attemptMs: PTY_MS, retries: 1 },
{ file: 'test/skill-e2e-plan-mode-no-op.test.ts', attemptMs: 5 * CAPTURE_LONG_MS, retries: 2 },
{ file: 'test/skill-e2e-plan-ceo-mode-routing.test.ts', attemptMs: 2 * CAPTURE_LONG_MS, retries: 1 },
{ file: 'test/skill-e2e-plan-eng-plan-mode.test.ts', attemptMs: 2 * CAPTURE_LONG_MS, retries: 1 },
{ file: 'test/skill-e2e-plan-prosons.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), retries: 1 },
// Gate: six 300s cases + one 610s case; periodic: two 900s + three 600s.
{ file: 'test/skill-e2e-plan.test.ts', attemptMs: Math.max(6 * CAPTURE_MS + CAPTURE_LONG_MS + 10_000, 2 * PTY_MS + 3 * CAPTURE_LONG_MS), retries: 1 },
].map(({ file, attemptMs, retries }) => ({
file, attemptMs, retries,
id: `${file.slice('test/'.length, -'.test.ts'.length)}-existing-retry-v1`,
shardReserveMs: AUTOPLAN_CHAIN_BUDGET.shardReserveMs,
shardMs: attemptMs * (retries + 1) + AUTOPLAN_CHAIN_BUDGET.shardReserveMs,
})),
];
/** The only registered over-tier test budget; arbitrary per-file escapes fail. */
export function assertPaidTestBudget(file: string, ms: number): void {
if (!Number.isSafeInteger(ms) || ms <= 0 ||
(ms > PTY_LONG_MS * 1.25 &&
(file !== AUTOPLAN_CHAIN_BUDGET.file || ms !== AUTOPLAN_CHAIN_BUDGET.testMs))) {
throw new Error(`Unregistered paid test budget: ${file}: ${ms}`);
}
}