v1.91.7.0 feat: add functional QA and pre-publication docs checks (#2983)

* feat: add surface-aware exploratory QA and ship documentation gates

* test: preserve delegated QA setup authority after main integration

* fix(qa): clarify exploration order and preserve report artifacts

* test(qa): follow the shared setup reference directly

* refactor(ship): make verification and recovery routes explicit

* test(ship): align evidence and review guards with explicit routes

* fix(workflows): clarify ship recovery and functional QA evidence

* fix(workflows): clarify approval recovery and full QA coverage

* refactor(workflows): order review transactions and clarify ship state

* fix(ship): clarify final verification and fail closed at publication

* fix(evals): attribute native atomic documentation writes

* fix(ship): clarify recovery and documentation lifecycle guidance

* fix(test): preserve observed native placeholder styling in CI

* fix(codex): report watchdog timeouts without a process-exit race

* Checkpoint functional QA implementation and workflow validation repairs

* Fix documentation and shared-review fixture contracts

* docs: clarify judge reuse and evaluation supervision

* test: align review evidence and selected case contracts

* test: verify append-only documentation checkpoints and recovery

* fix: qualify QA workflows and CI validation repairs

* fix: launch shared-libs fixture scripts on Windows

* fix: qualify QA deadlines, fixture isolation, and shard cleanup

* fix: preserve qualified QA and cancellation repairs

* fix: enforce functional fixture authority and share strict event decoding

* fix: retain free-test evidence and explain recovery

* fix: reject malformed native evidence after decoder consolidation

* test: use reliable capture for telemetry privacy filters

* test: refresh measured quick coverage and document validation costs

* Fix native fixture receipts and preserve VM validation evidence

* Align negative judge controls with upstream clarity policy

* Fix report-only QA preparation and public evidence handling

* Clarify QA-only preparation and current-report preservation

* Stream Ship quality judgments with an explicit 64k response contract

* Validate compact judge reasoning locally with supported wire schema

* Align functional QA fixture instructions with evidence acceptance

* Bind native browser diagnostics to execution evidence and align review verdicts

* Preserve native diagnostic line boundaries

* Serialize functional QA evidence from native captures

* Keep large QA evidence fixture payload out of Windows argv
This commit is contained in:
Garry Tan authored and GitHub committed 2026-09-29 06:07:35 -07:00
1 parent 65bfb0ce49
commit dcaea52800
333 files changed
+41755 -7357

No files matched your search

+36 -10
View File
@@ -1,6 +1,9 @@
import { describe, test, expect } from 'bun:test';
import * as fs from 'fs';
import * as path from 'path';
import { generateReviewDashboard } from '../scripts/resolvers/review';
import { HOST_PATHS } from '../scripts/resolvers/types';
import { ALL_HOST_CONFIGS } from '../hosts';
/**
* Template-drift tripwire for the content-binding wave. The bins are
@@ -39,14 +42,15 @@ describe('content-binding template drift', () => {
test('ship historical readiness does not replace the current pre-landing gate', () => {
const text = rendered('ship/SKILL.md');
expect(text).not.toContain('The only review that gates shipping');
expect(text).toContain('Step 9 remains mandatory');
expect(text).toContain('This verdict never skips Step 9 or its finding, approval and convergence gates');
});
test('ship Step 16 carries the evidence check (mechanized IRON LAW)', () => {
const ship = rendered('ship/SKILL.md');
expect(ship).toMatch(/gstack-evidence check --label tests --expect-cmd '[^']+' --label vitest --expect-cmd '[^']+' --max-age 24 --allow-paths CHANGELOG\.md,VERSION,package\.json/);
expect(ship).toContain('A failed CHECK identifies evidence to repair; it is not a test failure');
expect(ship).toContain('required live RUN must pass');
expect(ship.replace(/\s+/g, ' ')).toContain("| STALE/MISSING: changed content, command or age, or no proven run | Run `~/.claude/skills/gstack/bin/gstack-evidence run --label <lane> -- '<command>'`, read the result and recheck once");
expect(ship.replace(/\s+/g, ' ')).toContain("**New, changed or unwaived test failure:** STOP publication. Run Steps 5–15, starting with Step 5's triage, then return to Step 16 stage 1");
expect(ship).toContain('return to Step 16 stage 1');
});
test('ship Step 5 lanes run wrapped with per-lane labels', () => {
@@ -71,8 +75,8 @@ describe('content-binding template drift', () => {
// {{REVIEW_DASHBOARD}}; ship is the canonical carrier.
const ship = rendered('ship/SKILL.md');
expect(ship).toContain('---WTREE---');
expect(ship).toContain('diff-scoped rows only');
expect(ship).toContain('grade UNKNOWN and treat as stale');
expect(ship).toContain('Content-first rule');
expect(ship).toContain('A failed command means UNKNOWN, treated as stale');
});
test('the diff-scoped row list is IDENTICAL in both grading surfaces (no drift)', () => {
@@ -80,7 +84,7 @@ describe('content-binding template drift', () => {
// they diverged once (codex-review present in one, missing in the other).
// Rendered dashboards escape backticks (template-literal origin), so match
// structurally: the three row names in order inside the rule sentence.
const rowList = /diff-scoped rows only:[\s\S]{0,80}?adversarial-review[\s\S]{0,80}?codex-review[\s\S]{0,80}?ship-stage entries/;
const rowList = /Content-first rule[\s\S]{0,80}?`review`[\s\S]{0,80}?`adversarial-review`[\s\S]{0,80}?`codex-review`[\s\S]{0,80}?ship-stage (?:entries|reviews)[\s\S]{0,80}?`design-review-lite`/;
expect(rendered('ship/SKILL.md')).toMatch(rowList);
// land-and-deploy's copy of the row list lives in the carved readiness-gate
// section (Step 3.5a), not the skeleton.
@@ -93,8 +97,8 @@ describe('content-binding template drift', () => {
expect(text).toContain('review_freshness');
expect(text).toContain('UNVERIFIED');
expect(text).toContain('Never fall back');
expect(text).toContain('0 commits');
expect(text.toLowerCase()).toContain('plan-tier');
expect(text).toMatch(/(?:0|zero) commits/);
expect(text.toLowerCase()).toMatch(/plan-tier|plan records/);
}
});
@@ -106,7 +110,12 @@ describe('content-binding template drift', () => {
const army = rendered('ship/sections/review-army.md');
expect(army.indexOf('gstack-review-log --start review')).toBeLessThan(army.indexOf('run `git diff origin/<base>`'));
expect(army).toContain('--finish REVIEW_START');
expect(army).toContain('persist item 6 below with `converged:false`');
expect(army.replace(/\s+/g, ' ')).toContain('Complete items 5–6 exactly once with the original REVIEW_START');
expect(army).toContain('fixes also require `converged:false`');
const ship = rendered('ship/SKILL.md');
expect(army.replace(/\s+/g, ' ')).toContain('**Third fixing cycle reached (`CYCLES >= 3`):** STOP and report recurring findings with `converged:false`; do not run a fourth fixing cycle');
expect(ship.replace(/\s+/g, ' ')).toContain('Keep the same attempt counts throughout the invocation');
expect(ship.replace(/\s+/g, ' ')).toContain('a repair never resets approvals or expands them');
expect(army).toContain('--start design-review-lite');
expect(army).toContain('--finish DESIGN_START');
const codex = rendered('codex/sections/review-mode.md');
@@ -121,11 +130,28 @@ describe('content-binding template drift', () => {
const adversarial = rendered(`${skill}/sections/adversarial.md`);
expect(adversarial).toContain('--start adversarial-review');
expect(adversarial).toContain('--finish PASS_START');
expect(adversarial).toContain('Each outside adversarial/structured pass');
expect(adversarial.replace(/\s+/g, ' ')).toContain('Do the same before each outside adversarial or structured pass reads its diff');
expect(adversarial).toContain('Each token is consumed once');
}
});
test('dashboard selection and freshness precede a verdict without replacing the live ship gate', () => {
for (const host of ALL_HOST_CONFIGS) {
const text = generateReviewDashboard({ host: host.name, skillName: 'ship',
tmplPath: 'ship/SKILL.md.tmpl', paths: HOST_PATHS[host.name] }).replace(/\s+/g, ' ');
const positions = ['**1. Choose the records', '**2. Check freshness',
'**3. Choose the historical verdict', '**4. Display the dashboard'].map(marker => text.indexOf(marker));
expect(positions.every(position => position >= 0)).toBe(true);
expect(positions).toEqual([...positions].sort((a, b) => a - b));
expect(text).toContain('never substitute an older success for a newer failure');
expect(text).toContain('CLEARED requires the selected Eng Review to be `clean`, within 7 days and fresh under step 2');
expect(text).toContain('STALE or UNVERIFIED cannot clear Eng Review');
expect(text).toContain('Missing `review_freshness`, including legacy log-only records, means UNVERIFIED');
expect(text).toContain('This verdict never skips Step 9 or its finding, approval and convergence gates');
expect(text).toContain('Continue Step 1 even when history is NOT CLEARED');
}
});
test('release-body write side carries the banner tripwire (and it actually fires)', () => {
const body = rendered('document-release/sections/release-body.md');
expect(body).toContain('grep -c "UNTRUSTED TRACKER CONTENT" "<run-dir>/body.md"');