From 080a655791d4ac2af40b10bbfae8ae611df6d8f0 Mon Sep 17 00:00:00 2001 From: Garry Tan Date: Mon, 28 Sep 2026 16:00:12 -0700 Subject: [PATCH] v1.91.5.0 feat: balanced free-suite shards, test:ubicloud, and faster PR eval lane (#2989) * v1.91.5.0 feat: balanced free-suite shards, 16-way Linux runs, and bun run test:ubicloud * chore: regenerate agents digest for v1.91.5.0 * ci: serial flaky retry and flake ledger for the Windows free lane * ci: cancel superseded eval runs; relax LLM-judge clarity bar to 3 --- .github/workflows/evals-periodic.yml | 7 +- .github/workflows/evals.yml | 14 +- .github/workflows/quality-gate.yml | 3 + .github/workflows/windows-free-tests.yml | 16 + AGENTS.md | 1 + CHANGELOG.md | 18 + CLAUDE.md | 1 + CONTRIBUTING.md | 14 +- VERSION | 2 +- agents-digest/gstack-AGENTS.md | 2 +- browse/test/terminal-agent-lifecycle.test.ts | 11 +- design/test/variants-retry-after.test.ts | 8 +- docs/TESTING_INTERNALS.md | 57 +- package.json | 3 +- scripts/free-test-durations.json | 2040 +++++++++-------- scripts/test-free-shards.ts | 50 +- scripts/ubicloud/setup-free-suite.sh | 48 + scripts/ubicloud/test-free.sh | 21 + scripts/ubicloud/ubi-runner.sh | 282 +++ test/ci-paid-coordination.test.ts | 11 +- test/egress-receipt-wiring.test.ts | 2 + test/helpers/setup-codex-scope-fixture.ts | 132 ++ test/setup-codex-scope-migrations.test.ts | 264 +++ test/setup-codex-scope-namespaces.test.ts | 273 +++ ...etup-codex-scope-review-boundaries.test.ts | 87 + test/setup-codex-scope.test.ts | 736 +----- test/skill-llm-eval.test.ts | 22 +- test/test-free-shards-sandbox-knobs.test.ts | 53 +- test/test-free-shards.test.ts | 17 + test/ubicloud-runner.test.ts | 63 + 30 files changed, 2474 insertions(+), 1784 deletions(-) create mode 100755 scripts/ubicloud/setup-free-suite.sh create mode 100755 scripts/ubicloud/test-free.sh create mode 100755 scripts/ubicloud/ubi-runner.sh create mode 100644 test/helpers/setup-codex-scope-fixture.ts create mode 100644 test/setup-codex-scope-migrations.test.ts create mode 100644 test/setup-codex-scope-namespaces.test.ts create mode 100644 test/setup-codex-scope-review-boundaries.test.ts create mode 100644 test/ubicloud-runner.test.ts diff --git a/.github/workflows/evals-periodic.yml b/.github/workflows/evals-periodic.yml index 8cde616bf..9092b1633 100644 --- a/.github/workflows/evals-periodic.yml +++ b/.github/workflows/evals-periodic.yml @@ -275,9 +275,10 @@ jobs: report: runs-on: ubicloud-standard-2 needs: [plan-slices, eval-slices, gate-census] - # always(): the report must run (and FAIL) when an executor died — a - # missing slice artifact reading as green is the class this lane kills. - if: always() && needs.plan-slices.result == 'success' + # !cancelled(): the report must run (and FAIL) when an executor died — a + # missing slice artifact reading as green is the class this lane kills — + # but a cancelled run stops here. + if: ${{ !cancelled() && needs.plan-slices.result == 'success' }} timeout-minutes: 10 permissions: contents: read diff --git a/.github/workflows/evals.yml b/.github/workflows/evals.yml index 7e680c24d..0764d867b 100644 --- a/.github/workflows/evals.yml +++ b/.github/workflows/evals.yml @@ -178,7 +178,10 @@ jobs: eval-slices: runs-on: ubicloud-standard-8 needs: [build-image, plan-slices] - if: always() && needs.build-image.result == 'success' && needs.plan-slices.result == 'success' + # !cancelled(), not always(): still runs when build-image was skipped + # (image already published), but a newer push's cancel-in-progress stops + # it instead of letting a superseded run finish its paid slices first. + if: ${{ !cancelled() && needs.build-image.result == 'success' && needs.plan-slices.result == 'success' }} # Aggregate spawn-concurrency budget: 6 slices x EVALS_JOBS=2 x # EVALS_CONCURRENCY=2 = 24 concurrent tests lane-wide (the old matrix's # 40-way per row queued claude session STARTUP behind 39 siblings and ate @@ -314,9 +317,10 @@ jobs: slices-report: runs-on: ubicloud-standard-2 needs: [plan-slices, eval-slices] - # always(): the report must run (and FAIL) when an executor died — a - # missing slice artifact reading as green is the class this lane kills. - if: always() && needs.plan-slices.result == 'success' + # !cancelled(): the report must run (and FAIL) when an executor died — a + # missing slice artifact reading as green is the class this lane kills — + # but a run superseded by a newer push stops here. + if: ${{ !cancelled() && needs.plan-slices.result == 'success' }} timeout-minutes: 5 # contents:read ONLY — this job executes the PR-authored reconcile # runner from the PR checkout, so it @@ -382,7 +386,7 @@ jobs: slices-comment: runs-on: ubicloud-standard-2 needs: slices-report - if: always() && github.event_name == 'pull_request' && github.event.pull_request.head.repo.full_name == github.repository && needs.slices-report.result != 'skipped' + if: ${{ !cancelled() && github.event_name == 'pull_request' && github.event.pull_request.head.repo.full_name == github.repository && needs.slices-report.result != 'skipped' }} timeout-minutes: 5 permissions: pull-requests: write diff --git a/.github/workflows/quality-gate.yml b/.github/workflows/quality-gate.yml index 9168e832a..7b17963e3 100644 --- a/.github/workflows/quality-gate.yml +++ b/.github/workflows/quality-gate.yml @@ -93,3 +93,6 @@ jobs: scripts/build-app.sh scripts/write-version-files.sh browse/scripts/build-node-server.sh + scripts/ubicloud/ubi-runner.sh + scripts/ubicloud/setup-free-suite.sh + scripts/ubicloud/test-free.sh diff --git a/.github/workflows/windows-free-tests.yml b/.github/workflows/windows-free-tests.yml index 4615ee56c..035e621a2 100644 --- a/.github/workflows/windows-free-tests.yml +++ b/.github/workflows/windows-free-tests.yml @@ -146,6 +146,13 @@ jobs: # exclusions — don't resurrect a hand list in this file. env: GSTACK_FREE_JOBS: '2' + # Same policy as the required Linux lane: process-supervision timing + # tests pass alone on this 4-vCPU runner but can stall under the + # two-shard load, so attributed failures get one serial retry + # (cap 5 files, truncation veto). Every flaky pass is appended to + # the ledger uploaded below, so repeats stay visible and ranked. + GSTACK_FREE_RETRY_FLAKY: '1' + GSTACK_FLAKE_LEDGER: ${{ runner.temp }}/flake-ledger.jsonl # Point os.tmpdir() at the runner temp so the shard logs land # somewhere the artifact step below can glob. TEMP: ${{ runner.temp }} @@ -176,6 +183,15 @@ jobs: path: ${{ runner.temp }}/gstack-free-test-*.log if-no-files-found: ignore + - name: Upload flake ledger + if: always() + uses: actions/upload-artifact@v7 + with: + name: flake-ledger-windows + path: ${{ runner.temp }}/flake-ledger.jsonl + if-no-files-found: ignore + retention-days: 90 + cookie-native-qualification: if: github.event_name == 'workflow_dispatch' && !inputs.dia_native_only && !inputs.native_diagnostics_only && !inputs.dia_launch_comparison && !inputs.dia_gui_readiness runs-on: windows-latest diff --git a/AGENTS.md b/AGENTS.md index dbff6b9cf..ffa86c7ef 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -236,6 +236,7 @@ When fixing failures or preparing `/ship`, follow this order: bun install # install dependencies bun run test:quick # fast measured free subset for edit feedback (not acceptance) bun run test # complete free suite via the strict shard runner (no API spend) +bun run test:ubicloud # same suite on an ephemeral 16-vCPU Ubicloud VM (needs UBICLOUD_API_KEY) bun run eval:bg:pr # changed fast live probes + selected judges, with explicit deferrals bun run eval:bg:release # fresh complete gate + periodic live coverage bun run test:windows # curated Windows-safe subset (runs on windows-latest) diff --git a/CHANGELOG.md b/CHANGELOG.md index fa9c788fe..12ac9110e 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,23 @@ # Changelog +## [1.91.5.0] - 2026-09-28 + +The free suite now finishes in about half the time on a 16-core Linux machine, `bun run test:ubicloud` runs it on a fresh 16-vCPU Ubicloud VM from any dev box, container, or cloud sandbox, and re-pushing a PR no longer waits behind the previous commit's eval run. + +### Added +- `bun run test:ubicloud [test:free args]` runs the complete free suite on an ephemeral Ubicloud VM (`UBICLOUD_API_KEY` required; `UBI_SIZE` and `UBI_LOCATION` choose the machine). The VM gets the required CI lane's environment and strictness settings, uncommitted edits are included, shard logs are copied to `.context/ubicloud/`, and the VM is always destroyed afterwards. A full run takes about four and a half minutes end to end, compared with seven and a half minutes of suite time alone on a 4-vCPU machine. +- `bun run test:ubicloud --record-durations` refreshes `scripts/free-test-durations.json` in that same environment and copies it back. +- `scripts/ubicloud/ubi-runner.sh` runs any command on a Ubicloud VM (`run`, or `up` / `ssh` / `sync` / `pull` / `down`). It needs only bash, curl, python3, ssh, and tar locally, and it removes its own stale VMs after 12 hours. + +### Changed +- On Linux, `bun run test` now starts one shard per available CPU, up to 16; macOS and Windows keep the limit of six. On a 16-vCPU VM, 16 shards finished the suite in 137 seconds and six shards took 327 seconds. +- The duration seed is re-recorded with browser, display, and CSO tests actually running, and the slowest test file is split into four files. On a 16-vCPU run, all 16 shards now finish within about 15 seconds of each other, where one shard used to take twice as long as the rest. The 20 CI shards are packed with the same seed. +- Full-suite and CI planning runs name any test file missing from the duration seed, so a slow new file cannot quietly become the long pole. +- The Windows CI lane gets the same single serial retry for attributed failures as the required Linux lane, and uploads its flaky passes as a `flake-ledger-windows` artifact. Its process-supervision timing tests pass when run alone but can stall under the lane's two-shard load. +- A new push to a pull request now cancels the previous commit's paid eval run. The eval slice, report, and comment jobs run unless the workflow is cancelled (`!cancelled()`) instead of unconditionally (`always()`), so they still run when the image build is skipped or a slice fails, but a superseded run no longer finishes (and bills) its slices while the new commit's run waits behind it. A workflow test fails if any eval job goes back to a job-level `always()`. +- LLM-judge skill quality evals require a clarity score of 3 instead of 4; completeness and actionability bars are unchanged. The cookie setup judge keeps its manually approved thresholds. +- Two timing-sensitive tests allow for a heavily loaded machine: the terminal-agent startup race waits up to 8 seconds for a cold agent start, and the invalid Retry-After cases accept timer delays below the next 4-second backoff step. + ## [1.91.4.0] - 2026-09-28 Windows users can copy signed-in cookies from Opera and Opera GX into gstack's browser. On Windows, where Chrome, Edge and Brave increasingly store App-Bound Encryption cookies that gstack cannot decrypt, Opera and Opera GX still use DPAPI-protected cookies, so they may be the browsers where import keeps working. diff --git a/CLAUDE.md b/CLAUDE.md index b13916917..de8abebf8 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -6,6 +6,7 @@ bun install # install dependencies bun run test:quick # measured fast deterministic subset for edit feedback bun run test # complete free suite via the strict parallel runner +bun run test:ubicloud # complete free suite on an ephemeral 16-vCPU Ubicloud VM (needs UBICLOUD_API_KEY) bun run test:pr # changed fast live probes + selected judges (CI PR default) bun run test:evals # run paid evals: LLM judge + E2E (diff-based, ~$4.35/run max) bun run test:evals:all # run ALL paid evals regardless of diff diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 3830e6fe5..7679288b2 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -182,6 +182,7 @@ Bun auto-loads `.env` — no extra config. Conductor workspaces inherit `.env` f bun run test:quick # Measured fast free subset for ordinary edits; not full acceptance bun run eval:bg:pr # Changed fast live probes + selected quality judges, detached bun run test # Final full free acceptance after focused repairs and source freeze +bun run test:ubicloud # Same suite on an ephemeral 16-vCPU Ubicloud VM; needs UBICLOUD_API_KEY bun run test:e2e # Tier 2: E2E only (needs EVALS=1, can't run inside Claude Code) bun run test:evals # Tier 2 + 3 combined (~$4.35/run) ``` @@ -208,11 +209,15 @@ unknown-input results cannot be reused. Timing goals are under one minute for edit feedback, 3–5 minutes for typical PR checks, and 60–90 seconds for complete free test execution across isolated CI machines. They are targets, not timeout reductions or guarantees. The complete -local suite uses available CPU affinity, up to six workers; use `test:quick` for -the shorter edit loop. The historical six-worker result below and the +local suite uses available CPU affinity, up to 16 workers on Linux and six on +macOS and Windows; use `test:quick` for the shorter edit loop. On a small dev +box, container, or cloud sandbox, `bun run test:ubicloud` runs the complete suite +on a fresh 16-vCPU Ubicloud VM with the CI lane's environment instead (about +four and a half minutes end to end, including VM boot and setup). The +historical six-worker result below and the [four-CPU portfolio comparison](docs/TEST_PORTFOLIO.md#measurement-contract) are machine-specific measurements. CI setup, build and queue time are reported -separately. Refresh measurements with `bun run test:free --record-durations`; +separately. Refresh measurements with `bun run test:ubicloud --record-durations`; the required free CI lane packs the complete inventory across isolated runners, then checks every shard's receipt before reporting success. Local worker counts remain bounded to avoid browser/process contention. @@ -254,7 +259,8 @@ flakes; the required CI free lane turns it on and uploads every flaky pass in a JSONL ledger artifact that `bun run eval:flake-rank` folds in). Working in a cloud sandbox? Run `scripts/sandbox-doctor.sh` once per boot to make the suite run green (details in -[docs/TESTING_INTERNALS.md](docs/TESTING_INTERNALS.md)). +[docs/TESTING_INTERNALS.md](docs/TESTING_INTERNALS.md)), or skip the sandbox's +limits entirely with `bun run test:ubicloud`. Don't type bare `bun test` for the suite: it walks the whole repo, loads paid eval files, and misses the strict classifier. No API keys needed. diff --git a/VERSION b/VERSION index 52de62705..4a05f8036 100644 --- a/VERSION +++ b/VERSION @@ -1 +1 @@ -1.91.4.0 +1.91.5.0 diff --git a/agents-digest/gstack-AGENTS.md b/agents-digest/gstack-AGENTS.md index a0448edf0..d1275f30a 100644 --- a/agents-digest/gstack-AGENTS.md +++ b/agents-digest/gstack-AGENTS.md @@ -1,4 +1,4 @@ -# gstack digest v1.91.4.0 — regenerate/re-copy after upgrading gstack +# gstack digest v1.91.5.0 — regenerate/re-copy after upgrading gstack Behavioral rules from gstack (https://github.com/garrytan/gstack), compressed for agent hosts without a full skill install. The full skills add workflows, diff --git a/browse/test/terminal-agent-lifecycle.test.ts b/browse/test/terminal-agent-lifecycle.test.ts index 1b71a8bbc..2d36b6c7d 100644 --- a/browse/test/terminal-agent-lifecycle.test.ts +++ b/browse/test/terminal-agent-lifecycle.test.ts @@ -311,15 +311,18 @@ describe('terminal-agent owned lifecycle regression', () => { try { writeAgentRecord(stateDir, { pid: old.pid, gen: 'synthetic-old-generation', startedAt: Date.now(), startTime: readAgentStartTime(old.pid), ownerPid: process.pid, ownerStartTime }); - expect(await waitFor(() => fs.existsSync(`${barrier}.ready`))).toBe(true); + // Startup waits cover a cold `bun run` of the agent; under a fully + // loaded 16-shard run that alone can exceed 3s. They return as soon as + // the file appears, so the ordering contract below is unchanged. + expect(await waitFor(() => fs.existsSync(`${barrier}.ready`), 8000)).toBe(true); winner = rawAgent('synthetic-new-generation', false); writeAgentRecord(stateDir, { pid: winner.pid, gen: 'synthetic-new-generation', startedAt: Date.now(), startTime: readAgentStartTime(winner.pid), ownerPid: process.pid, ownerStartTime }); - expect(await waitFor(() => fs.existsSync(path.join(stateDir, 'terminal-port')))).toBe(true); + expect(await waitFor(() => fs.existsSync(path.join(stateDir, 'terminal-port')), 8000)).toBe(true); const port = fs.readFileSync(path.join(stateDir, 'terminal-port'), 'utf8'); const token = fs.readFileSync(path.join(stateDir, 'terminal-internal-token'), 'utf8'); fs.writeFileSync(barrier, 'continue'); - expect(await Promise.race([old.exited.then(() => true), Bun.sleep(3000).then(() => false)])).toBe(true); + expect(await Promise.race([old.exited.then(() => true), Bun.sleep(8000).then(() => false)])).toBe(true); expect(readAgentRecord(stateDir)?.gen).toBe('synthetic-new-generation'); expect(fs.readFileSync(path.join(stateDir, 'terminal-port'), 'utf8')).toBe(port); expect(fs.readFileSync(path.join(stateDir, 'terminal-internal-token'), 'utf8')).toBe(token); @@ -330,7 +333,7 @@ describe('terminal-agent owned lifecycle regression', () => { await old.exited; if (winner) await winner.exited; } - }, 10000); + }, 30000); test('daemon respawns after agent crash, then exits without deleting a successor state', async () => { const stateDir = dir(); diff --git a/design/test/variants-retry-after.test.ts b/design/test/variants-retry-after.test.ts index 801c3e944..fa2d31839 100644 --- a/design/test/variants-retry-after.test.ts +++ b/design/test/variants-retry-after.test.ts @@ -108,9 +108,11 @@ describe("generateVariant Retry-After handling", () => { expect(result.success).toBe(true); expect(calls.length).toBe(2); const gap = calls[1].ts - calls[0].ts; - // Falls through to existing 2s exponential leading delay + // Falls through to existing 2s exponential leading delay. The ceiling + // only has to reject the next backoff step (4s) or a double wait; timers + // on a loaded scheduler fire over a second late. expect(gap).toBeGreaterThanOrEqual(1800); - expect(gap).toBeLessThan(3000); + expect(gap).toBeLessThan(3900); }); test("no Retry-After header: falls through to exponential", async () => { @@ -125,7 +127,7 @@ describe("generateVariant Retry-After handling", () => { expect(calls.length).toBe(2); const gap = calls[1].ts - calls[0].ts; expect(gap).toBeGreaterThanOrEqual(1800); - expect(gap).toBeLessThan(3000); + expect(gap).toBeLessThan(3900); }); test("Retry-After: 0 retries immediately, skips leading exponential", async () => { diff --git a/docs/TESTING_INTERNALS.md b/docs/TESTING_INTERNALS.md index 9a32e75b1..285987793 100644 --- a/docs/TESTING_INTERNALS.md +++ b/docs/TESTING_INTERNALS.md @@ -161,14 +161,20 @@ load-sensitive on a busy dev box, runs only in CI or on explicit opt-in **Free suite (`bun run test:free`).** `scripts/test-free-shards.ts` runs N concurrent shard processes (serial within each) with strict-output -classification per shard. Local defaults use the available CPU affinity, -floored at one and capped at six; `GSTACK_FREE_JOBS` remains an explicit override. +classification per shard. Local defaults use the available CPU affinity +(which Bun also bounds by a container's cgroup CPU quota), floored at one and +capped at 16 on Linux and six on macOS and Windows; `GSTACK_FREE_JOBS` remains +an explicit override. This does not change the separate CI machine count. Full-suite shards are packed by RECORDED PER-FILE DURATIONS (LPT, `packShardsByDuration`) when the committed seed -`scripts/free-test-durations.json` exists — refresh it occasionally with -`bun run test:free --record-durations` (each file timed in its own child; -CI never records). Missing seed → silent hash-shard fallback; corrupt seed → -one warning + fallback; unknown files get 75th-percentile pessimism. Packed +`scripts/free-test-durations.json` exists — refresh it with +`bun run test:ubicloud --record-durations`, which times each file in its own +child on a VM with the CI lane's environment and copies the seed back (CI never +records; a seed recorded where browser or display tests skip underestimates +them). Missing seed → silent hash-shard fallback; corrupt seed → one warning + +fallback; unknown files get 75th-percentile pessimism, and both full-suite and +`--ci-plan` runs name them on stderr so a new slow file cannot silently become +the long pole. Packed shards get duration-aware walls (`max(base, predicted × 3, files × 5s)`). The legacy `--shards N --shard i` path keeps stable hash indices. Required CI uses one duration-packed `--ci-plan`, 20 isolated `--ci-run` machines, and a @@ -256,7 +262,7 @@ serially. Plans and receipts bind the source revision, complete inventory and strict outcomes; missing, duplicate or mismatched receipts fail the required aggregate. The existing maximum five-file flaky retry allowance applies across the entire lane, not separately to every machine. Refresh the full timing list -with `bun run test:free --record-durations`. Profiling records failures faithfully +with `bun run test:ubicloud --record-durations`. Profiling records failures faithfully and is separate from final release acceptance. **CI planner/executor/report.** `--emit-plan --slices K` computes @@ -388,6 +394,39 @@ against a temp `GSTACK_INSTALL_DIR` / `GSTACK_SKILLS_DIR`, and `freeze/bin/check-freeze.sh` with JSON payloads on stdin (including the `GSTACK_HOME` state-root parity against `bin/gstack-paths`). +## Ubicloud VMs (`bun run test:ubicloud`) + +`bun run test:ubicloud [test:free args]` runs the free suite on an ephemeral +Ubicloud VM instead of the local machine. It is the fast path from small dev +boxes, containers, and cloud sandboxes, and the way to record the duration seed +in the CI lane's environment. It needs `UBICLOUD_API_KEY` (a project token from +the Ubicloud console) plus `bash`, `curl`, `python3`, `ssh`, `ssh-keygen`, and +`tar` locally. + +`scripts/ubicloud/ubi-runner.sh` creates the VM (`UBI_SIZE`, default +`standard-16`; `UBI_LOCATION`, default `eu-central-h1`) and streams the +checkout to it: tracked files, untracked files that are not ignored, and +`.git`, so uncommitted edits are tested. It then runs +`scripts/ubicloud/setup-free-suite.sh`, which mirrors the `free-suite` CI job +(same Bun pin, Playwright Chromium with its setuid sandbox helper, Xvfb, +poppler, emoji fonts, generated host outputs, gate binaries, and the CSO +helper), and runs `xvfb-run -a bun run test:free` with `GSTACK_EXPECT_BINARIES=1` +and `GSTACK_FREE_RETRY_FLAKY=1`. Shard logs are copied to +`.context/ubicloud//`, and the VM is destroyed on every exit path. +The exit status is the suite's. + +A stock Ubuntu 24.04 VM differs from a GitHub-hosted runner in three ways that +the scripts correct: the login umask is `002` (group-writable directories fail +the CSO private-state checks), AppArmor blocks the unprivileged user +namespaces Chromium's sandbox needs, and `clang` and `python3-venv` are absent +(required by the Dia readiness and Python runner tests). + +VMs are named `ubirun--`. Every new VM first destroys `ubirun-*` +VMs older than `UBI_GC_HOURS` (default 12), so an interrupted client cannot +leak one for long. For other commands, use the runner directly: +`scripts/ubicloud/ubi-runner.sh run --setup