v1.91.7.0 feat: add functional QA and pre-publication docs checks (#2983)

* feat: add surface-aware exploratory QA and ship documentation gates

* test: preserve delegated QA setup authority after main integration

* fix(qa): clarify exploration order and preserve report artifacts

* test(qa): follow the shared setup reference directly

* refactor(ship): make verification and recovery routes explicit

* test(ship): align evidence and review guards with explicit routes

* fix(workflows): clarify ship recovery and functional QA evidence

* fix(workflows): clarify approval recovery and full QA coverage

* refactor(workflows): order review transactions and clarify ship state

* fix(ship): clarify final verification and fail closed at publication

* fix(evals): attribute native atomic documentation writes

* fix(ship): clarify recovery and documentation lifecycle guidance

* fix(test): preserve observed native placeholder styling in CI

* fix(codex): report watchdog timeouts without a process-exit race

* Checkpoint functional QA implementation and workflow validation repairs

* Fix documentation and shared-review fixture contracts

* docs: clarify judge reuse and evaluation supervision

* test: align review evidence and selected case contracts

* test: verify append-only documentation checkpoints and recovery

* fix: qualify QA workflows and CI validation repairs

* fix: launch shared-libs fixture scripts on Windows

* fix: qualify QA deadlines, fixture isolation, and shard cleanup

* fix: preserve qualified QA and cancellation repairs

* fix: enforce functional fixture authority and share strict event decoding

* fix: retain free-test evidence and explain recovery

* fix: reject malformed native evidence after decoder consolidation

* test: use reliable capture for telemetry privacy filters

* test: refresh measured quick coverage and document validation costs

* Fix native fixture receipts and preserve VM validation evidence

* Align negative judge controls with upstream clarity policy

* Fix report-only QA preparation and public evidence handling

* Clarify QA-only preparation and current-report preservation

* Stream Ship quality judgments with an explicit 64k response contract

* Validate compact judge reasoning locally with supported wire schema

* Align functional QA fixture instructions with evidence acceptance

* Bind native browser diagnostics to execution evidence and align review verdicts

* Preserve native diagnostic line boundaries

* Serialize functional QA evidence from native captures

* Keep large QA evidence fixture payload out of Windows argv
This commit is contained in:
Garry Tan authored and GitHub committed 2026-09-29 06:07:35 -07:00
1 parent 65bfb0ce49
commit dcaea52800
333 files changed
+41755 -7357

No files matched your search

+44 -11
View File
@@ -4,7 +4,7 @@ name: Periodic Evals
# tests can't rot invisibly — the class where the autoplan-dual-voice E2E was
# silently broken for months until a lucky local diff selected it. Engine:
# scripts/test-paid-shards.ts (the same runner local eval:bg:periodic uses):
# one planner manifest, 6 ordinary slices plus overlay and Autoplan slices, and a FAIL-CLOSED report — a slice
# one planner manifest, 7 ordinary slices plus overlay and Autoplan slices, and a FAIL-CLOSED report — a slice
# whose artifact never landed is a failure, not an absence. The gate-census
# job is the weekly EVALS_ALL backstop for the gate tier (PR lanes are
# diff-billed, so without it the full gate census might never execute
@@ -96,7 +96,7 @@ jobs:
- name: Emit run manifest (ALL periodic tests minus reasoned excludes)
env:
EVALS_ALL: "1"
run: EVALS_TIER=periodic bun --no-install run scripts/test-paid-shards.ts --tier periodic --emit-plan /tmp/paid-plan/manifest.json --slices 8 --autoplan-slice
run: EVALS_TIER=periodic bun --no-install run scripts/test-paid-shards.ts --tier periodic --emit-plan /tmp/paid-plan/manifest.json --slices 9 --autoplan-slice
- uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
@@ -107,7 +107,7 @@ jobs:
- name: Emit gate census manifest (ALL gate tests)
env:
EVALS_ALL: "1"
run: EVALS_TIER=gate bun run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/gate-census-plan/manifest.json --slices 7
run: EVALS_TIER=gate bun run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/gate-census-plan/manifest.json --slices 8
- uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
@@ -118,9 +118,11 @@ jobs:
eval-slices:
runs-on: ubicloud-standard-8
needs: [build-image, plan-slices]
# Eight slices retain every registered case and retry. The complete
# census needs at most 338 minutes per slice, plus 20 minutes setup/upload.
timeout-minutes: 358
env:
EVALS_RUN_ID: ci-${{ github.run_id }}-${{ github.run_attempt }}-eval-slices-${{ matrix.slice }}
# Nine slices retain every registered case and retry. The complete
# census needs at most 292m20 per slice, plus 20 minutes setup/upload.
timeout-minutes: 360
permissions:
contents: read
packages: read
@@ -132,8 +134,9 @@ jobs:
options: --user runner
strategy:
fail-fast: false
max-parallel: 8
matrix:
slice: [1, 2, 3, 4, 5, 6, 7, 8]
slice: [1, 2, 3, 4, 5, 6, 7, 8, 9]
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
with:
@@ -171,7 +174,7 @@ jobs:
name: paid-plan
path: /tmp/paid-plan
- name: Run slice ${{ matrix.slice }}/8
- name: Run slice ${{ matrix.slice }}/9
env:
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
@@ -190,6 +193,20 @@ jobs:
path: /tmp/paid-slice-results
retention-days: 90
- name: Upload native capture evidence
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: native-captures-${{ env.EVALS_RUN_ID }}
include-hidden-files: true
path: |
~/.gstack/projects/*/e2e-runs
~/.gstack/projects/*/evals/qa-callers
~/.gstack-dev/e2e-runs
~/.gstack-dev/evals/qa-callers
if-no-files-found: ignore
retention-days: 90
- name: Upload shard logs on failure
if: failure()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
@@ -212,7 +229,9 @@ jobs:
gate-census:
runs-on: ubicloud-standard-8
needs: [build-image, plan-slices]
# Seven slices need at most 304m each, plus 20 minutes setup/upload.
env:
EVALS_RUN_ID: ci-${{ github.run_id }}-${{ github.run_attempt }}-gate-census-${{ matrix.slice }}
# Eight slices need at most 302m each, plus 20 minutes setup/upload.
timeout-minutes: 352
permissions:
contents: read
@@ -228,7 +247,7 @@ jobs:
fail-fast: false
max-parallel: 4
matrix:
slice: [1, 2, 3, 4, 5, 6, 7]
slice: [1, 2, 3, 4, 5, 6, 7, 8]
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
with:
@@ -253,7 +272,7 @@ jobs:
name: gate-census-plan
path: /tmp/gate-census-plan
- name: Run gate census slice ${{ matrix.slice }}/7
- name: Run gate census slice ${{ matrix.slice }}/8
env:
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
@@ -272,6 +291,20 @@ jobs:
path: /tmp/gate-census-results
retention-days: 90
- name: Upload native capture evidence
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: native-captures-${{ env.EVALS_RUN_ID }}
include-hidden-files: true
path: |
~/.gstack/projects/*/e2e-runs
~/.gstack/projects/*/evals/qa-callers
~/.gstack-dev/e2e-runs
~/.gstack-dev/evals/qa-callers
if-no-files-found: ignore
retention-days: 90
report:
runs-on: ubicloud-standard-2
needs: [plan-slices, eval-slices, gate-census]