mirror of
https://github.com/garrytan/gstack.git
synced 2026-10-02 17:40:02 +02:00
v1.91.7.0 feat: add functional QA and pre-publication docs checks (#2983)
* feat: add surface-aware exploratory QA and ship documentation gates * test: preserve delegated QA setup authority after main integration * fix(qa): clarify exploration order and preserve report artifacts * test(qa): follow the shared setup reference directly * refactor(ship): make verification and recovery routes explicit * test(ship): align evidence and review guards with explicit routes * fix(workflows): clarify ship recovery and functional QA evidence * fix(workflows): clarify approval recovery and full QA coverage * refactor(workflows): order review transactions and clarify ship state * fix(ship): clarify final verification and fail closed at publication * fix(evals): attribute native atomic documentation writes * fix(ship): clarify recovery and documentation lifecycle guidance * fix(test): preserve observed native placeholder styling in CI * fix(codex): report watchdog timeouts without a process-exit race * Checkpoint functional QA implementation and workflow validation repairs * Fix documentation and shared-review fixture contracts * docs: clarify judge reuse and evaluation supervision * test: align review evidence and selected case contracts * test: verify append-only documentation checkpoints and recovery * fix: qualify QA workflows and CI validation repairs * fix: launch shared-libs fixture scripts on Windows * fix: qualify QA deadlines, fixture isolation, and shard cleanup * fix: preserve qualified QA and cancellation repairs * fix: enforce functional fixture authority and share strict event decoding * fix: retain free-test evidence and explain recovery * fix: reject malformed native evidence after decoder consolidation * test: use reliable capture for telemetry privacy filters * test: refresh measured quick coverage and document validation costs * Fix native fixture receipts and preserve VM validation evidence * Align negative judge controls with upstream clarity policy * Fix report-only QA preparation and public evidence handling * Clarify QA-only preparation and current-report preservation * Stream Ship quality judgments with an explicit 64k response contract * Validate compact judge reasoning locally with supported wire schema * Align functional QA fixture instructions with evidence acceptance * Bind native browser diagnostics to execution evidence and align review verdicts * Preserve native diagnostic line boundaries * Serialize functional QA evidence from native captures * Keep large QA evidence fixture payload out of Windows argv
This commit is contained in:
1 parent
65bfb0ce49
commit
dcaea52800
333 files changed
+41755
-7357
No files matched your search
@@ -4,7 +4,7 @@ name: Periodic Evals
|
||||
# tests can't rot invisibly — the class where the autoplan-dual-voice E2E was
|
||||
# silently broken for months until a lucky local diff selected it. Engine:
|
||||
# scripts/test-paid-shards.ts (the same runner local eval:bg:periodic uses):
|
||||
# one planner manifest, 6 ordinary slices plus overlay and Autoplan slices, and a FAIL-CLOSED report — a slice
|
||||
# one planner manifest, 7 ordinary slices plus overlay and Autoplan slices, and a FAIL-CLOSED report — a slice
|
||||
# whose artifact never landed is a failure, not an absence. The gate-census
|
||||
# job is the weekly EVALS_ALL backstop for the gate tier (PR lanes are
|
||||
# diff-billed, so without it the full gate census might never execute
|
||||
@@ -96,7 +96,7 @@ jobs:
|
||||
- name: Emit run manifest (ALL periodic tests minus reasoned excludes)
|
||||
env:
|
||||
EVALS_ALL: "1"
|
||||
run: EVALS_TIER=periodic bun --no-install run scripts/test-paid-shards.ts --tier periodic --emit-plan /tmp/paid-plan/manifest.json --slices 8 --autoplan-slice
|
||||
run: EVALS_TIER=periodic bun --no-install run scripts/test-paid-shards.ts --tier periodic --emit-plan /tmp/paid-plan/manifest.json --slices 9 --autoplan-slice
|
||||
|
||||
- uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
with:
|
||||
@@ -107,7 +107,7 @@ jobs:
|
||||
- name: Emit gate census manifest (ALL gate tests)
|
||||
env:
|
||||
EVALS_ALL: "1"
|
||||
run: EVALS_TIER=gate bun run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/gate-census-plan/manifest.json --slices 7
|
||||
run: EVALS_TIER=gate bun run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/gate-census-plan/manifest.json --slices 8
|
||||
|
||||
- uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
with:
|
||||
@@ -118,9 +118,11 @@ jobs:
|
||||
eval-slices:
|
||||
runs-on: ubicloud-standard-8
|
||||
needs: [build-image, plan-slices]
|
||||
# Eight slices retain every registered case and retry. The complete
|
||||
# census needs at most 338 minutes per slice, plus 20 minutes setup/upload.
|
||||
timeout-minutes: 358
|
||||
env:
|
||||
EVALS_RUN_ID: ci-${{ github.run_id }}-${{ github.run_attempt }}-eval-slices-${{ matrix.slice }}
|
||||
# Nine slices retain every registered case and retry. The complete
|
||||
# census needs at most 292m20 per slice, plus 20 minutes setup/upload.
|
||||
timeout-minutes: 360
|
||||
permissions:
|
||||
contents: read
|
||||
packages: read
|
||||
@@ -132,8 +134,9 @@ jobs:
|
||||
options: --user runner
|
||||
strategy:
|
||||
fail-fast: false
|
||||
max-parallel: 8
|
||||
matrix:
|
||||
slice: [1, 2, 3, 4, 5, 6, 7, 8]
|
||||
slice: [1, 2, 3, 4, 5, 6, 7, 8, 9]
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
with:
|
||||
@@ -171,7 +174,7 @@ jobs:
|
||||
name: paid-plan
|
||||
path: /tmp/paid-plan
|
||||
|
||||
- name: Run slice ${{ matrix.slice }}/8
|
||||
- name: Run slice ${{ matrix.slice }}/9
|
||||
env:
|
||||
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
|
||||
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
|
||||
@@ -190,6 +193,20 @@ jobs:
|
||||
path: /tmp/paid-slice-results
|
||||
retention-days: 90
|
||||
|
||||
- name: Upload native capture evidence
|
||||
if: always()
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
with:
|
||||
name: native-captures-${{ env.EVALS_RUN_ID }}
|
||||
include-hidden-files: true
|
||||
path: |
|
||||
~/.gstack/projects/*/e2e-runs
|
||||
~/.gstack/projects/*/evals/qa-callers
|
||||
~/.gstack-dev/e2e-runs
|
||||
~/.gstack-dev/evals/qa-callers
|
||||
if-no-files-found: ignore
|
||||
retention-days: 90
|
||||
|
||||
- name: Upload shard logs on failure
|
||||
if: failure()
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
@@ -212,7 +229,9 @@ jobs:
|
||||
gate-census:
|
||||
runs-on: ubicloud-standard-8
|
||||
needs: [build-image, plan-slices]
|
||||
# Seven slices need at most 304m each, plus 20 minutes setup/upload.
|
||||
env:
|
||||
EVALS_RUN_ID: ci-${{ github.run_id }}-${{ github.run_attempt }}-gate-census-${{ matrix.slice }}
|
||||
# Eight slices need at most 302m each, plus 20 minutes setup/upload.
|
||||
timeout-minutes: 352
|
||||
permissions:
|
||||
contents: read
|
||||
@@ -228,7 +247,7 @@ jobs:
|
||||
fail-fast: false
|
||||
max-parallel: 4
|
||||
matrix:
|
||||
slice: [1, 2, 3, 4, 5, 6, 7]
|
||||
slice: [1, 2, 3, 4, 5, 6, 7, 8]
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
with:
|
||||
@@ -253,7 +272,7 @@ jobs:
|
||||
name: gate-census-plan
|
||||
path: /tmp/gate-census-plan
|
||||
|
||||
- name: Run gate census slice ${{ matrix.slice }}/7
|
||||
- name: Run gate census slice ${{ matrix.slice }}/8
|
||||
env:
|
||||
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
|
||||
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
|
||||
@@ -272,6 +291,20 @@ jobs:
|
||||
path: /tmp/gate-census-results
|
||||
retention-days: 90
|
||||
|
||||
- name: Upload native capture evidence
|
||||
if: always()
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
with:
|
||||
name: native-captures-${{ env.EVALS_RUN_ID }}
|
||||
include-hidden-files: true
|
||||
path: |
|
||||
~/.gstack/projects/*/e2e-runs
|
||||
~/.gstack/projects/*/evals/qa-callers
|
||||
~/.gstack-dev/e2e-runs
|
||||
~/.gstack-dev/evals/qa-callers
|
||||
if-no-files-found: ignore
|
||||
retention-days: 90
|
||||
|
||||
report:
|
||||
runs-on: ubicloud-standard-2
|
||||
needs: [plan-slices, eval-slices, gate-census]
|
||||
|
||||
@@ -141,7 +141,7 @@ jobs:
|
||||
if: github.event_name != 'workflow_dispatch' || inputs.validation_phase == 'all'
|
||||
env:
|
||||
EVALS_ALL: ${{ (github.event_name == 'workflow_dispatch' && inputs.evals_all) && '1' || '' }}
|
||||
run: EVALS_TIER=gate bun --no-install run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/paid-plan/manifest.json --slices 6
|
||||
run: EVALS_TIER=gate bun --no-install run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/paid-plan/manifest.json --slices 7
|
||||
|
||||
- name: Emit validation-phase manifest
|
||||
if: github.event_name == 'workflow_dispatch' && inputs.validation_phase != 'all'
|
||||
@@ -178,6 +178,8 @@ jobs:
|
||||
eval-slices:
|
||||
runs-on: ubicloud-standard-8
|
||||
needs: [build-image, plan-slices]
|
||||
env:
|
||||
EVALS_RUN_ID: ci-${{ github.run_id }}-${{ github.run_attempt }}-eval-slices-${{ matrix.slice }}
|
||||
# !cancelled(), not always(): still runs when build-image was skipped
|
||||
# (image already published), but a newer push's cancel-in-progress stops
|
||||
# it instead of letting a superseded run finish its paid slices first.
|
||||
@@ -187,9 +189,9 @@ jobs:
|
||||
# 40-way per row queued claude session STARTUP behind 39 siblings and ate
|
||||
# per-test budgets — the documented timeout-flake family). Tune with
|
||||
# parity data before raising.
|
||||
# The complete gate census needs at most 236 minutes per slice; keep
|
||||
# The complete gate census needs at most 242 minutes per slice; keep
|
||||
# 20 minutes for setup/upload without preempting configured retries.
|
||||
timeout-minutes: 256
|
||||
timeout-minutes: 265
|
||||
permissions:
|
||||
contents: read
|
||||
packages: read
|
||||
@@ -201,8 +203,9 @@ jobs:
|
||||
options: --user runner
|
||||
strategy:
|
||||
fail-fast: false
|
||||
max-parallel: 6
|
||||
matrix:
|
||||
slice: [1, 2, 3, 4, 5, 6]
|
||||
slice: [1, 2, 3, 4, 5, 6, 7]
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
with:
|
||||
@@ -251,7 +254,7 @@ jobs:
|
||||
key: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-${{ github.run_id }}-${{ github.run_attempt }}-${{ matrix.slice }}
|
||||
restore-keys: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-
|
||||
|
||||
- name: Run slice ${{ matrix.slice }}/6
|
||||
- name: Run slice ${{ matrix.slice }}/7
|
||||
env:
|
||||
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
|
||||
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
|
||||
@@ -297,6 +300,20 @@ jobs:
|
||||
path: /tmp/paid-slice-results
|
||||
retention-days: 90
|
||||
|
||||
- name: Upload native capture evidence
|
||||
if: always()
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
with:
|
||||
name: native-captures-${{ env.EVALS_RUN_ID }}
|
||||
include-hidden-files: true
|
||||
path: |
|
||||
~/.gstack/projects/*/e2e-runs
|
||||
~/.gstack/projects/*/evals/qa-callers
|
||||
~/.gstack-dev/e2e-runs
|
||||
~/.gstack-dev/evals/qa-callers
|
||||
if-no-files-found: ignore
|
||||
retention-days: 90
|
||||
|
||||
# The spooled per-shard full logs — a red weekly/PR lane three weeks
|
||||
# later needs more than a summary line.
|
||||
- name: Upload shard logs on failure
|
||||
|
||||
@@ -295,7 +295,8 @@ jobs:
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
with:
|
||||
name: free-test-shard-logs-${{ matrix.shard }}
|
||||
path: /tmp/gstack-free-test-*.log
|
||||
path: .context/free-test-logs/gstack-free-test-*.log
|
||||
include-hidden-files: true
|
||||
if-no-files-found: ignore
|
||||
|
||||
# Branch protection already requires the `free-tests` context. Keep that
|
||||
|
||||
@@ -153,8 +153,6 @@ jobs:
|
||||
# the ledger uploaded below, so repeats stay visible and ranked.
|
||||
GSTACK_FREE_RETRY_FLAKY: '1'
|
||||
GSTACK_FLAKE_LEDGER: ${{ runner.temp }}/flake-ledger.jsonl
|
||||
# Point os.tmpdir() at the runner temp so the shard logs land
|
||||
# somewhere the artifact step below can glob.
|
||||
TEMP: ${{ runner.temp }}
|
||||
TMP: ${{ runner.temp }}
|
||||
run: bun run test:windows
|
||||
@@ -180,7 +178,10 @@ jobs:
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
name: windows-free-test-shard-logs
|
||||
path: ${{ runner.temp }}/gstack-free-test-*.log
|
||||
path: |
|
||||
.context/free-test-logs/gstack-free-test-*.log
|
||||
${{ runner.temp }}/gstack-free-test-*.log
|
||||
include-hidden-files: true
|
||||
if-no-files-found: ignore
|
||||
|
||||
- name: Upload flake ledger
|
||||
|
||||
Reference in new issue
Block a user