v1.91.7.0 feat: add functional QA and pre-publication docs checks (#2983)

* feat: add surface-aware exploratory QA and ship documentation gates

* test: preserve delegated QA setup authority after main integration

* fix(qa): clarify exploration order and preserve report artifacts

* test(qa): follow the shared setup reference directly

* refactor(ship): make verification and recovery routes explicit

* test(ship): align evidence and review guards with explicit routes

* fix(workflows): clarify ship recovery and functional QA evidence

* fix(workflows): clarify approval recovery and full QA coverage

* refactor(workflows): order review transactions and clarify ship state

* fix(ship): clarify final verification and fail closed at publication

* fix(evals): attribute native atomic documentation writes

* fix(ship): clarify recovery and documentation lifecycle guidance

* fix(test): preserve observed native placeholder styling in CI

* fix(codex): report watchdog timeouts without a process-exit race

* Checkpoint functional QA implementation and workflow validation repairs

* Fix documentation and shared-review fixture contracts

* docs: clarify judge reuse and evaluation supervision

* test: align review evidence and selected case contracts

* test: verify append-only documentation checkpoints and recovery

* fix: qualify QA workflows and CI validation repairs

* fix: launch shared-libs fixture scripts on Windows

* fix: qualify QA deadlines, fixture isolation, and shard cleanup

* fix: preserve qualified QA and cancellation repairs

* fix: enforce functional fixture authority and share strict event decoding

* fix: retain free-test evidence and explain recovery

* fix: reject malformed native evidence after decoder consolidation

* test: use reliable capture for telemetry privacy filters

* test: refresh measured quick coverage and document validation costs

* Fix native fixture receipts and preserve VM validation evidence

* Align negative judge controls with upstream clarity policy

* Fix report-only QA preparation and public evidence handling

* Clarify QA-only preparation and current-report preservation

* Stream Ship quality judgments with an explicit 64k response contract

* Validate compact judge reasoning locally with supported wire schema

* Align functional QA fixture instructions with evidence acceptance

* Bind native browser diagnostics to execution evidence and align review verdicts

* Preserve native diagnostic line boundaries

* Serialize functional QA evidence from native captures

* Keep large QA evidence fixture payload out of Windows argv
This commit is contained in:
Garry Tan authored and GitHub committed 2026-09-29 06:07:35 -07:00
1 parent 65bfb0ce49
commit dcaea52800
333 files changed
+41755 -7357

No files matched your search

+99
View File
@@ -0,0 +1,99 @@
import { describe, expect, test } from 'bun:test';
import { readFileSync } from 'node:fs';
import { ALL_HOST_CONFIGS } from '../hosts';
import { generateAdversarialStep, generateCrossReviewDedup, generateSharedCodeReuse } from '../scripts/resolvers/review';
import { HOST_PATHS } from '../scripts/resolvers/types';
const compact = (text: string) => text.replace(/\s+/g, ' ');
const review = compact(readFileSync(new URL('../ship/sections/review-army.md.tmpl', import.meta.url), 'utf8'));
describe.each(ALL_HOST_CONFIGS.map(({ name }) => name))('%s ship skip/requeue contract', host => {
const ctx = { host, skillName: 'ship', tmplPath: '', paths: HOST_PATHS[host] };
const dedup = compact(generateCrossReviewDedup(ctx));
const adversarial = compact(generateAdversarialStep(ctx));
const finish = adversarial.slice(adversarial.indexOf('### Finish the adversarial phase'));
test('unchanged explicit skips are matched before the actionable queue drives a repeat', () => {
const match = finish.indexOf("Apply Step 9.3's matching procedure");
expect(match).toBeGreaterThan(-1);
expect(match).toBeLessThan(finish.indexOf('2. **Fixes queued'));
expect(finish).toContain('Only unmatched or reopened findings remain queued');
expect(dedup).toContain('Only explicit `skipped` actions qualify');
expect(dedup).toContain('If both history and the invocation action list lack decisions, classify normally');
expect(dedup).toContain('Revalidated Skips suppress repeat questions and fixes');
expect(dedup).toContain('Report the suppressed count once if nonzero');
expect(dedup).toContain('reopen the finding; unrelated edits do not');
const steps = ['1. **Validate severity.**', '2. **Read decisions.**',
'3. **Match evidence.**', '4. **Match shared-code structurally.**', '5. **Apply dispositions.**'];
const positions = steps.map(step => dedup.indexOf(step));
expect(positions.every(position => position >= 0)).toBe(true);
expect(positions).toEqual([...positions].sort((a, b) => a - b));
expect(review).toContain('Classify only unmatched or reopened findings as AUTO-FIX or ASK');
expect(review).toContain('after Step 9.3 matches all sources, including queued Steps 10–11 findings');
});
test('dedup and persistence include queued sources and preserve the explicit decision', () => {
expect(dedup).toContain('exploratory QA and queued Steps 10–11 findings');
expect(review).toContain('checklist, specialist, exploratory QA and queued Steps 10–11 records');
expect(review).toContain('Save each explicit Skip immediately in the invocation action list');
expect(review).toContain('preserve `advisory`, `evidence_paths` and `helper_target`');
});
test('working-tree changes and new evidence reopen matching identities', () => {
expect(dedup).toContain('git diff --name-only <prior-review-commit>');
expect(dedup).not.toContain('git diff --name-only <prior-review-commit> HEAD');
expect(dedup).toContain('committed, staged, unstaged and non-ignored untracked source');
expect(dedup).toContain('Changed inputs, proposal, behavior, risk or new evidence reopen the finding');
expect(dedup).toContain('honoring later user decisions');
expect(dedup).toContain('same fingerprint, advisory/defect kind and scope');
expect(dedup).toContain('Compare supporting source and finding evidence with the saved decision');
expect(dedup).toContain('as a shortlist, not proof');
});
test('fixed regressions and missing Skip proof are not suppressed', () => {
expect(dedup).toContain('never `fixed`, `auto-fixed` or unanswered questions');
expect(dedup).toContain('Missing proof or unknown comparisons require a fresh decision, not suppression');
expect(finish).toContain('Keep scoped approvals');
expect(dedup).toContain('Require the same fingerprint, advisory/defect kind and scope');
expect(finish).toContain('Unvalidated historical Skips stay unmatched for the full Step 9 repeat below');
expect(finish).toContain('never jump to 9.3 or mint a late REVIEW_START');
});
test('shared-code and advisory collisions retain stricter identity checks', () => {
expect(dedup).toContain('they cannot suppress defects');
expect(dedup).toContain('remove `advisory`, never downgrade severity');
expect(dedup).toContain('Reject contradictory saved decisions');
expect(dedup).toContain('requires re-reading all callers (including indirect callers) and the helper destination');
expect(dedup).toContain('Missing metadata never permits ordinary line matching');
expect(dedup).toContain('Prior-review reuse additionally requires the checker below; invocation decisions cannot replace it');
const checker = compact(generateSharedCodeReuse(ctx));
expect(checker).toContain('Only `reusable: true` permits suppression');
expect(checker).toContain('False, command failure or unreadable output requires fresh source review');
expect(checker).toContain('Do not supply your own snapshot, prior record or coverage');
});
test('skipped defects stay unresolved and required failures stay failed', () => {
expect(dedup).toContain('not unresolved defects: retain them in counts, status and the final report');
expect(dedup).toContain('Keep required-probe failures failed');
expect(review).toContain('Skipping a fix is not risk acceptance or a passing probe');
expect(review).toContain('Failed, blocked, inconclusive or not-run required probes mean false, never clean');
expect(review).toContain('VERIFY_RESULT stays fail');
expect(adversarial).toContain('retain the acknowledged findings and failed gate; do not report a clean review');
});
test('fresh review after edits, native coverage and cycle limits remain required', () => {
expect(finish).toContain('**Required native review incomplete:** STOP');
expect(finish).toContain('Insert Steps 9, 10 and 11 before the pending Step 11.5');
expect(finish).toContain("never resets Step 9's three-cycle fix limit");
expect(review).toContain('Set CYCLES to 0 on first entry only');
expect(review).toContain('Increment CYCLES once if fixes were applied');
expect(review).toContain('do not run a fourth fixing cycle');
expect(finish).toContain('then continue to Step 11.5. Never jump directly to release preparation');
});
});
test('ship invocation matching does not change standalone review routing', () => {
const ctx = { host: 'claude' as const, skillName: 'review', tmplPath: '', paths: HOST_PATHS.claude };
expect(generateCrossReviewDedup(ctx)).not.toContain('5. **Apply dispositions.**');
expect(generateAdversarialStep(ctx)).not.toContain('### Finish the adversarial phase');
});