v1.91.7.0 feat: add functional QA and pre-publication docs checks (#2983)

* feat: add surface-aware exploratory QA and ship documentation gates

* test: preserve delegated QA setup authority after main integration

* fix(qa): clarify exploration order and preserve report artifacts

* test(qa): follow the shared setup reference directly

* refactor(ship): make verification and recovery routes explicit

* test(ship): align evidence and review guards with explicit routes

* fix(workflows): clarify ship recovery and functional QA evidence

* fix(workflows): clarify approval recovery and full QA coverage

* refactor(workflows): order review transactions and clarify ship state

* fix(ship): clarify final verification and fail closed at publication

* fix(evals): attribute native atomic documentation writes

* fix(ship): clarify recovery and documentation lifecycle guidance

* fix(test): preserve observed native placeholder styling in CI

* fix(codex): report watchdog timeouts without a process-exit race

* Checkpoint functional QA implementation and workflow validation repairs

* Fix documentation and shared-review fixture contracts

* docs: clarify judge reuse and evaluation supervision

* test: align review evidence and selected case contracts

* test: verify append-only documentation checkpoints and recovery

* fix: qualify QA workflows and CI validation repairs

* fix: launch shared-libs fixture scripts on Windows

* fix: qualify QA deadlines, fixture isolation, and shard cleanup

* fix: preserve qualified QA and cancellation repairs

* fix: enforce functional fixture authority and share strict event decoding

* fix: retain free-test evidence and explain recovery

* fix: reject malformed native evidence after decoder consolidation

* test: use reliable capture for telemetry privacy filters

* test: refresh measured quick coverage and document validation costs

* Fix native fixture receipts and preserve VM validation evidence

* Align negative judge controls with upstream clarity policy

* Fix report-only QA preparation and public evidence handling

* Clarify QA-only preparation and current-report preservation

* Stream Ship quality judgments with an explicit 64k response contract

* Validate compact judge reasoning locally with supported wire schema

* Align functional QA fixture instructions with evidence acceptance

* Bind native browser diagnostics to execution evidence and align review verdicts

* Preserve native diagnostic line boundaries

* Serialize functional QA evidence from native captures

* Keep large QA evidence fixture payload out of Windows argv
This commit is contained in:
Garry Tan authored and GitHub committed 2026-09-29 06:07:35 -07:00
1 parent 65bfb0ce49
commit dcaea52800
333 files changed
+41755 -7357

No files matched your search

+44 -11
View File
@@ -4,7 +4,7 @@ name: Periodic Evals
# tests can't rot invisibly — the class where the autoplan-dual-voice E2E was
# silently broken for months until a lucky local diff selected it. Engine:
# scripts/test-paid-shards.ts (the same runner local eval:bg:periodic uses):
# one planner manifest, 6 ordinary slices plus overlay and Autoplan slices, and a FAIL-CLOSED report — a slice
# one planner manifest, 7 ordinary slices plus overlay and Autoplan slices, and a FAIL-CLOSED report — a slice
# whose artifact never landed is a failure, not an absence. The gate-census
# job is the weekly EVALS_ALL backstop for the gate tier (PR lanes are
# diff-billed, so without it the full gate census might never execute
@@ -96,7 +96,7 @@ jobs:
- name: Emit run manifest (ALL periodic tests minus reasoned excludes)
env:
EVALS_ALL: "1"
run: EVALS_TIER=periodic bun --no-install run scripts/test-paid-shards.ts --tier periodic --emit-plan /tmp/paid-plan/manifest.json --slices 8 --autoplan-slice
run: EVALS_TIER=periodic bun --no-install run scripts/test-paid-shards.ts --tier periodic --emit-plan /tmp/paid-plan/manifest.json --slices 9 --autoplan-slice
- uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
@@ -107,7 +107,7 @@ jobs:
- name: Emit gate census manifest (ALL gate tests)
env:
EVALS_ALL: "1"
run: EVALS_TIER=gate bun run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/gate-census-plan/manifest.json --slices 7
run: EVALS_TIER=gate bun run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/gate-census-plan/manifest.json --slices 8
- uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
@@ -118,9 +118,11 @@ jobs:
eval-slices:
runs-on: ubicloud-standard-8
needs: [build-image, plan-slices]
# Eight slices retain every registered case and retry. The complete
# census needs at most 338 minutes per slice, plus 20 minutes setup/upload.
timeout-minutes: 358
env:
EVALS_RUN_ID: ci-${{ github.run_id }}-${{ github.run_attempt }}-eval-slices-${{ matrix.slice }}
# Nine slices retain every registered case and retry. The complete
# census needs at most 292m20 per slice, plus 20 minutes setup/upload.
timeout-minutes: 360
permissions:
contents: read
packages: read
@@ -132,8 +134,9 @@ jobs:
options: --user runner
strategy:
fail-fast: false
max-parallel: 8
matrix:
slice: [1, 2, 3, 4, 5, 6, 7, 8]
slice: [1, 2, 3, 4, 5, 6, 7, 8, 9]
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
with:
@@ -171,7 +174,7 @@ jobs:
name: paid-plan
path: /tmp/paid-plan
- name: Run slice ${{ matrix.slice }}/8
- name: Run slice ${{ matrix.slice }}/9
env:
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
@@ -190,6 +193,20 @@ jobs:
path: /tmp/paid-slice-results
retention-days: 90
- name: Upload native capture evidence
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: native-captures-${{ env.EVALS_RUN_ID }}
include-hidden-files: true
path: |
~/.gstack/projects/*/e2e-runs
~/.gstack/projects/*/evals/qa-callers
~/.gstack-dev/e2e-runs
~/.gstack-dev/evals/qa-callers
if-no-files-found: ignore
retention-days: 90
- name: Upload shard logs on failure
if: failure()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
@@ -212,7 +229,9 @@ jobs:
gate-census:
runs-on: ubicloud-standard-8
needs: [build-image, plan-slices]
# Seven slices need at most 304m each, plus 20 minutes setup/upload.
env:
EVALS_RUN_ID: ci-${{ github.run_id }}-${{ github.run_attempt }}-gate-census-${{ matrix.slice }}
# Eight slices need at most 302m each, plus 20 minutes setup/upload.
timeout-minutes: 352
permissions:
contents: read
@@ -228,7 +247,7 @@ jobs:
fail-fast: false
max-parallel: 4
matrix:
slice: [1, 2, 3, 4, 5, 6, 7]
slice: [1, 2, 3, 4, 5, 6, 7, 8]
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
with:
@@ -253,7 +272,7 @@ jobs:
name: gate-census-plan
path: /tmp/gate-census-plan
- name: Run gate census slice ${{ matrix.slice }}/7
- name: Run gate census slice ${{ matrix.slice }}/8
env:
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
@@ -272,6 +291,20 @@ jobs:
path: /tmp/gate-census-results
retention-days: 90
- name: Upload native capture evidence
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: native-captures-${{ env.EVALS_RUN_ID }}
include-hidden-files: true
path: |
~/.gstack/projects/*/e2e-runs
~/.gstack/projects/*/evals/qa-callers
~/.gstack-dev/e2e-runs
~/.gstack-dev/evals/qa-callers
if-no-files-found: ignore
retention-days: 90
report:
runs-on: ubicloud-standard-2
needs: [plan-slices, eval-slices, gate-census]
+22 -5
View File
@@ -141,7 +141,7 @@ jobs:
if: github.event_name != 'workflow_dispatch' || inputs.validation_phase == 'all'
env:
EVALS_ALL: ${{ (github.event_name == 'workflow_dispatch' && inputs.evals_all) && '1' || '' }}
run: EVALS_TIER=gate bun --no-install run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/paid-plan/manifest.json --slices 6
run: EVALS_TIER=gate bun --no-install run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/paid-plan/manifest.json --slices 7
- name: Emit validation-phase manifest
if: github.event_name == 'workflow_dispatch' && inputs.validation_phase != 'all'
@@ -178,6 +178,8 @@ jobs:
eval-slices:
runs-on: ubicloud-standard-8
needs: [build-image, plan-slices]
env:
EVALS_RUN_ID: ci-${{ github.run_id }}-${{ github.run_attempt }}-eval-slices-${{ matrix.slice }}
# !cancelled(), not always(): still runs when build-image was skipped
# (image already published), but a newer push's cancel-in-progress stops
# it instead of letting a superseded run finish its paid slices first.
@@ -187,9 +189,9 @@ jobs:
# 40-way per row queued claude session STARTUP behind 39 siblings and ate
# per-test budgets — the documented timeout-flake family). Tune with
# parity data before raising.
# The complete gate census needs at most 236 minutes per slice; keep
# The complete gate census needs at most 242 minutes per slice; keep
# 20 minutes for setup/upload without preempting configured retries.
timeout-minutes: 256
timeout-minutes: 265
permissions:
contents: read
packages: read
@@ -201,8 +203,9 @@ jobs:
options: --user runner
strategy:
fail-fast: false
max-parallel: 6
matrix:
slice: [1, 2, 3, 4, 5, 6]
slice: [1, 2, 3, 4, 5, 6, 7]
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
with:
@@ -251,7 +254,7 @@ jobs:
key: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-${{ github.run_id }}-${{ github.run_attempt }}-${{ matrix.slice }}
restore-keys: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-
- name: Run slice ${{ matrix.slice }}/6
- name: Run slice ${{ matrix.slice }}/7
env:
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
@@ -297,6 +300,20 @@ jobs:
path: /tmp/paid-slice-results
retention-days: 90
- name: Upload native capture evidence
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: native-captures-${{ env.EVALS_RUN_ID }}
include-hidden-files: true
path: |
~/.gstack/projects/*/e2e-runs
~/.gstack/projects/*/evals/qa-callers
~/.gstack-dev/e2e-runs
~/.gstack-dev/evals/qa-callers
if-no-files-found: ignore
retention-days: 90
# The spooled per-shard full logs — a red weekly/PR lane three weeks
# later needs more than a summary line.
- name: Upload shard logs on failure
+2 -1
View File
@@ -295,7 +295,8 @@ jobs:
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: free-test-shard-logs-${{ matrix.shard }}
path: /tmp/gstack-free-test-*.log
path: .context/free-test-logs/gstack-free-test-*.log
include-hidden-files: true
if-no-files-found: ignore
# Branch protection already requires the `free-tests` context. Keep that
+4 -3
View File
@@ -153,8 +153,6 @@ jobs:
# the ledger uploaded below, so repeats stay visible and ranked.
GSTACK_FREE_RETRY_FLAKY: '1'
GSTACK_FLAKE_LEDGER: ${{ runner.temp }}/flake-ledger.jsonl
# Point os.tmpdir() at the runner temp so the shard logs land
# somewhere the artifact step below can glob.
TEMP: ${{ runner.temp }}
TMP: ${{ runner.temp }}
run: bun run test:windows
@@ -180,7 +178,10 @@ jobs:
uses: actions/upload-artifact@v7
with:
name: windows-free-test-shard-logs
path: ${{ runner.temp }}/gstack-free-test-*.log
path: |
.context/free-test-logs/gstack-free-test-*.log
${{ runner.temp }}/gstack-free-test-*.log
include-hidden-files: true
if-no-files-found: ignore
- name: Upload flake ledger