v1.91.7.0 feat: add functional QA and pre-publication docs checks (#2983)

* feat: add surface-aware exploratory QA and ship documentation gates

* test: preserve delegated QA setup authority after main integration

* fix(qa): clarify exploration order and preserve report artifacts

* test(qa): follow the shared setup reference directly

* refactor(ship): make verification and recovery routes explicit

* test(ship): align evidence and review guards with explicit routes

* fix(workflows): clarify ship recovery and functional QA evidence

* fix(workflows): clarify approval recovery and full QA coverage

* refactor(workflows): order review transactions and clarify ship state

* fix(ship): clarify final verification and fail closed at publication

* fix(evals): attribute native atomic documentation writes

* fix(ship): clarify recovery and documentation lifecycle guidance

* fix(test): preserve observed native placeholder styling in CI

* fix(codex): report watchdog timeouts without a process-exit race

* Checkpoint functional QA implementation and workflow validation repairs

* Fix documentation and shared-review fixture contracts

* docs: clarify judge reuse and evaluation supervision

* test: align review evidence and selected case contracts

* test: verify append-only documentation checkpoints and recovery

* fix: qualify QA workflows and CI validation repairs

* fix: launch shared-libs fixture scripts on Windows

* fix: qualify QA deadlines, fixture isolation, and shard cleanup

* fix: preserve qualified QA and cancellation repairs

* fix: enforce functional fixture authority and share strict event decoding

* fix: retain free-test evidence and explain recovery

* fix: reject malformed native evidence after decoder consolidation

* test: use reliable capture for telemetry privacy filters

* test: refresh measured quick coverage and document validation costs

* Fix native fixture receipts and preserve VM validation evidence

* Align negative judge controls with upstream clarity policy

* Fix report-only QA preparation and public evidence handling

* Clarify QA-only preparation and current-report preservation

* Stream Ship quality judgments with an explicit 64k response contract

* Validate compact judge reasoning locally with supported wire schema

* Align functional QA fixture instructions with evidence acceptance

* Bind native browser diagnostics to execution evidence and align review verdicts

* Preserve native diagnostic line boundaries

* Serialize functional QA evidence from native captures

* Keep large QA evidence fixture payload out of Windows argv
This commit is contained in:
Garry Tan authored and GitHub committed 2026-09-29 06:07:35 -07:00
1 parent 65bfb0ce49
commit dcaea52800
333 files changed
+41755 -7357

No files matched your search

+76
View File
@@ -0,0 +1,76 @@
import { expect, test } from 'bun:test';
import { readFileSync } from 'node:fs';
import { generatePlanCompletionGateShip } from '../scripts/resolvers/review';
import { generateTestBootstrap } from '../scripts/resolvers/testing';
import { HOST_PATHS } from '../scripts/resolvers/types';
const read = (file: string) => readFileSync(new URL(`../${file}`, import.meta.url), 'utf8');
const compact = (text: string) => text.replace(/\s+/g, ' ');
const ctx = { host: 'claude' as const, skillName: 'ship', tmplPath: '', paths: HOST_PATHS.claude };
test('ship STOP blocks advancement while retaining the stated repair route', () => {
const entry = compact(read('ship/SKILL.md.tmpl'));
expect(entry).toContain('STOP blocks advancement until the stated repair/resume route clears; without one, end this attempt');
expect(entry).toContain('Answer each AskUserQuestion before continuing');
const review = compact(read('ship/sections/review-army.md.tmpl'));
expect(review).toContain('**Dispatched reviewer output missing:** STOP');
expect(review).toContain('Retain queued fixes and restore coverage');
expect(review).toContain('**Third fixing cycle reached (`CYCLES >= 3`):** STOP');
expect(entry).toContain('Routine authorization never waives those gates or their required user decisions');
});
test('a new ship bootstrap choice overrides only the saved decline, not framework selection', () => {
const ship = generateTestBootstrap(ctx);
const decline = ship.slice(ship.indexOf('**If BOOTSTRAP_DECLINED**'), ship.indexOf('**If NO ecosystem marker matched:**'));
expect(compact(decline)).toContain("Step 5's explicit Add tests choice overrides that marker for this invocation only");
expect(compact(decline)).toContain('continue to runtime detection and B2–B3, including framework approval');
expect(decline).not.toContain('rm ');
expect(ship.indexOf('**If ANY existing-test evidence appears**')).toBeLessThan(ship.indexOf('**If BOOTSTRAP_DECLINED**'));
const qa = generateTestBootstrap({ ...ctx, skillName: 'qa' });
expect(qa).toContain('**If BOOTSTRAP_DECLINED** appears: Print "Test bootstrap previously declined — skipping." **Skip the rest of bootstrap.**');
expect(qa).not.toContain("Step 5's explicit Add tests choice");
});
test('an unverified item answered not done enters the existing decision before continuing', () => {
const gate = generatePlanCompletionGateShip(ctx);
const exits = compact(gate.slice(gate.indexOf('**Exit conditions:**'), gate.indexOf('**Cap.**')));
expect(exits).toContain('Any N: STOP and report that item as NOT DONE');
expect(exits).toContain('Resume only after its required work is verified');
expect(exits).toContain('no second deferral choice');
expect(exits).not.toContain('re-running /ship');
expect(gate).toContain('Per-item confirmation is mandatory');
expect(gate).toContain('with the user\'s free-text evidence');
});
test('docs reentry distinguishes the initial audit from same-invocation accepted evidence', () => {
const docs = compact(read('ship/sections/documentation.md.tmpl'));
const entry = docs.slice(0, docs.indexOf('## Prepare the candidate'));
expect(entry).toContain('First entry always launches the initial audit');
expect(entry).toContain('On reentry, reuse only this invocation\'s validated audit or named-risk decision');
expect(entry).toContain('Reentry never resets the count or authorizes a launch');
expect(entry).toContain("this invocation's validated audit or named-risk decision");
expect(entry).toContain('base/input hashes still match');
expect(entry).toContain('retain its actual status and scope');
expect(entry).toContain('Otherwise use Blocked recovery, not an unconditional launch');
expect(docs).toContain('never a third attempt, even after Step 16 changes');
expect(docs).toContain('never reuse an audit across invocations');
expect(docs).toContain('Unconfirmed writers, ownership violations, unauthorized Git mutation and redaction/security gates cannot be waived');
});
test('late behavioral repairs rebuild before the docs decision and commit only remaining changes', () => {
const entry = compact(read('ship/SKILL.md.tmpl'));
expect(entry).toContain('A range ending at Step 14 does not enter Step 14.5');
expect(entry).toContain('rebuild and compare again before stage 3 decides documentation freshness');
const commit = entry.slice(entry.indexOf('### 5. Report, then push'), entry.indexOf('## Step 17:'));
expect(commit).toContain('left uncommitted after Step 15');
expect(commit).toContain('never create an empty commit');
expect(commit).toContain('Preserve unrelated user files');
});
test('the title is prefixed once before the exact scanned value is published', () => {
const body = compact(read('ship/sections/pr-body.md.tmpl'));
expect(body).toContain("Use Step 18's `NEW_TITLE` unchanged; its version prefix is already present");
expect(body).not.toContain('`NEW_TITLE`, prefixed with');
expect(body).toContain('--title "$NEW_TITLE"');
expect(body).toContain('the same scanned `NEW_TITLE`');
});