v1.91.7.0 feat: add functional QA and pre-publication docs checks (#2983)

* feat: add surface-aware exploratory QA and ship documentation gates

* test: preserve delegated QA setup authority after main integration

* fix(qa): clarify exploration order and preserve report artifacts

* test(qa): follow the shared setup reference directly

* refactor(ship): make verification and recovery routes explicit

* test(ship): align evidence and review guards with explicit routes

* fix(workflows): clarify ship recovery and functional QA evidence

* fix(workflows): clarify approval recovery and full QA coverage

* refactor(workflows): order review transactions and clarify ship state

* fix(ship): clarify final verification and fail closed at publication

* fix(evals): attribute native atomic documentation writes

* fix(ship): clarify recovery and documentation lifecycle guidance

* fix(test): preserve observed native placeholder styling in CI

* fix(codex): report watchdog timeouts without a process-exit race

* Checkpoint functional QA implementation and workflow validation repairs

* Fix documentation and shared-review fixture contracts

* docs: clarify judge reuse and evaluation supervision

* test: align review evidence and selected case contracts

* test: verify append-only documentation checkpoints and recovery

* fix: qualify QA workflows and CI validation repairs

* fix: launch shared-libs fixture scripts on Windows

* fix: qualify QA deadlines, fixture isolation, and shard cleanup

* fix: preserve qualified QA and cancellation repairs

* fix: enforce functional fixture authority and share strict event decoding

* fix: retain free-test evidence and explain recovery

* fix: reject malformed native evidence after decoder consolidation

* test: use reliable capture for telemetry privacy filters

* test: refresh measured quick coverage and document validation costs

* Fix native fixture receipts and preserve VM validation evidence

* Align negative judge controls with upstream clarity policy

* Fix report-only QA preparation and public evidence handling

* Clarify QA-only preparation and current-report preservation

* Stream Ship quality judgments with an explicit 64k response contract

* Validate compact judge reasoning locally with supported wire schema

* Align functional QA fixture instructions with evidence acceptance

* Bind native browser diagnostics to execution evidence and align review verdicts

* Preserve native diagnostic line boundaries

* Serialize functional QA evidence from native captures

* Keep large QA evidence fixture payload out of Windows argv
This commit is contained in:
Garry Tan authored and GitHub committed 2026-09-29 06:07:35 -07:00
1 parent 65bfb0ce49
commit dcaea52800
333 files changed
+41755 -7357

No files matched your search

+130 -21
View File
@@ -3,9 +3,91 @@ import * as fs from 'fs';
import * as path from 'path';
import * as os from 'node:os';
import { spawnSync } from 'node:child_process';
import { generatePlanCompletionAuditReview, generatePlanCompletionAuditShip, generatePlanCompletionGateShip, generatePlanVerificationExec } from '../scripts/resolvers/review';
import { HOST_PATHS } from '../scripts/resolvers/types';
const SHIP_DIR = path.join(__dirname, '..', 'ship');
describe('authored ship-only plan verification handoff', () => {
const ctx = { skillName: 'ship', tmplPath: 'ship/SKILL.md.tmpl', host: 'claude' as const, paths: HOST_PATHS.claude };
test('local execution-only checks remain required in Step 8.1/9 outside implementation counts', () => {
const audit = generatePlanCompletionAuditShip(ctx);
const extraction = audit.slice(audit.indexOf('### Actionable Item Extraction'), audit.indexOf('### Verification Mode'));
expect(extraction).toContain('Separate deliverables from execution-only verification');
expect(extraction).toContain('implementation and test-creation requirements');
expect(extraction).toContain('retain its command, expected outcome and source verbatim');
expect(extraction).toContain('Step 8.1/9');
expect(extraction).toContain('outside implementation counts');
expect(extraction).toContain('never DONE from static inspection');
expect(extraction).toContain('not EXTERNAL-STATE merely because it has not run');
expect(extraction).toContain('Keep genuine external-state and human-only checks in this audit');
expect(extraction).toContain('zero implementation counts do not waive those checks');
expect(fs.readFileSync(path.join(SHIP_DIR, 'sections/plan-completion.md.tmpl'), 'utf8')).toContain('exactly these seven fields');
const gate = generatePlanCompletionGateShip(ctx);
expect(gate).toContain('Any NOT DONE items');
expect(gate).toContain('Per-item confirmation is mandatory');
});
test('review-mode extraction retains its existing categories and has no ship-only routing', () => {
const audit = generatePlanCompletionAuditReview({ ...ctx, skillName: 'review' });
expect(audit).toContain('**Test requirements:** "Test that X", "Add test for Y", "Verify Z"');
expect(audit).not.toContain('Separate deliverables from execution-only verification');
expect(audit).not.toContain('Step 8.1/9');
expect(audit).toContain('EXTERNAL-STATE');
});
test('preparation hands every explicit check to the report-only execution owner before Fix-First', () => {
const text = generatePlanVerificationExec(ctx).replace(/\s+/g, ' ');
expect(text).toContain('Collect now; execute in Step 9');
expect(text).toContain('Do not invoke an entire QA skill or start probes here');
for (const heading of ['Verification', 'Test plan', 'Testing', 'How to test', 'Manual testing']) {
expect(text).toContain(`\`${heading}\``);
}
expect(text).toContain('any other explicit checks, including execution-only items retained by Step 8');
expect(text).toContain('each exact expected outcome, source, surface, probe and safe prerequisites');
expect(text).toContain('Clarify unknown outcomes');
expect(text).toContain('Browser items use the declared project/plan dev URL and browser setup at execution');
expect(text).toContain('functional items use native tools without discovering a web server');
expect(text).toContain('An API URL is not automatically a page');
expect(text).toContain('Only browser evidence needs screenshots');
expect(text).toContain('If no verification section or no plan file exists, record no plan-specific items');
expect(text).toContain('Automatic diff-scoped QA still runs. Continue to Step 8.2 Scope Drift below');
expect(text).toContain('Handoff to Step 9.2.1');
expect(text).toContain('parent-owned report-only explorer must execute this complete list before Fix-First');
expect(text).toContain('prerequisite, permission, evidence and changed-input revalidation rules');
expect(text).toContain('Share current-input proof for overlapping smoke probes');
expect(text).toContain('plan checks beyond that smoke budget remain required');
expect(text).toContain('At command/time limits, mark remaining checks not run');
expect(text).toContain("Send failed, blocked or unrun checks through Step 9's required-probe gate, never silently waive them");
expect(text).toContain('Noninteractive runs return blocked');
expect(text).toContain('VERIFY_RESULT=pass only if all selected items pass, skipped only if none exist, otherwise fail');
expect(text).toContain('Risk acceptance keeps the actual failed, blocked and unrun outcomes');
expect(text).toContain("Report per-status counts, evidence and accepted risks in Step 19's `## Verification Results`, separately from automatic QA");
expect(text.indexOf('Collect now')).toBeLessThan(text.indexOf('Handoff to Step 9.2.1'));
expect(text.indexOf('Handoff to Step 9.2.1')).toBeLessThan(text.indexOf('After execution'));
});
test('the authored parent validates all seven fields and settles failed children before fallback', () => {
const audit = fs.readFileSync(path.join(SHIP_DIR, 'sections/plan-completion.md.tmpl'), 'utf8');
const parent = audit.slice(audit.indexOf('**Parent processing:**')).replace(/\s+/g, ' ');
expect(parent).toContain("Check the task's terminal status");
expect(parent).toContain('Without successful completion and valid LAST-line JSON, use the audit-failure fallback');
expect(parent).toContain('exactly the seven declared fields');
expect(parent).toContain('nonnegative integer counts whose classification sum equals `total_items`');
expect(parent).toContain('a string `summary`');
expect(parent).toContain('Missing, extra or invalid fields fail');
expect(parent).toContain('Valid no-plan/no-actionable reports retain zero counts and their summary');
expect(parent).toContain('no final output after ~10 minutes');
expect(parent).toContain('stop any live child and confirm it stopped before an inline audit');
expect(parent).toContain('same extraction/classification logic; never race a late result');
expect(parent).toContain('If that also fails, AskUserQuestion');
expect(parent).toContain('recording the reason in the PR body and Step 20 metrics');
expect(parent).toContain('Stop and fix the audit (recommended/default)');
const contract = audit.split('\n').find(line => line.startsWith('{"total_items":N,'))!;
expect(Object.keys(JSON.parse(contract.replace(/:N([,}])/g, ':0$1'))).sort())
.toEqual(['total_items', 'done', 'changed', 'partial', 'not_done', 'unverifiable', 'summary'].sort());
expect(generatePlanCompletionGateShip(ctx)).toContain('Only PARTIAL items (no NOT DONE, no UNVERIFIABLE)');
expect(generatePlanCompletionGateShip(ctx)).toContain('Continue with a note in the PR body. Not blocking');
});
});
// Carved (v2 plan T9): the Plan Completion gate moved into sections/plan-completion.md.
// Read the skeleton + sections union so these invariants follow the content.
function readShipUnion(): string {
@@ -42,18 +124,33 @@ describe('ship/SKILL.md — Plan Completion gate invariants (VAS-449 remediation
test('Subagent failure: fail-closed, not silent fail-open', () => {
expect(skill).not.toMatch(/Never block \/ship on subagent failure\.\s*$/m);
expect(skill).toMatch(/Silent fail-open is the failure shape that VAS-449 surfaced/);
expect(skill.replace(/\s+/g, ' ')).toContain('Silent fail-open is the failure shape that VAS-449 surfaced');
expect(skill).toMatch(/Stop and fix the audit/);
});
test('parent rejects audit errors and malformed counts instead of treating them as no plan', () => {
const audit = fs.readFileSync(path.join(SHIP_DIR, 'sections/plan-completion.md'), 'utf8');
const parent = audit.slice(audit.indexOf('**Parent processing:**'), audit.indexOf('**If the subagent fails'));
expect(parent).toContain('non-null `error`');
const parent = audit.slice(audit.indexOf('**Parent processing:**'), audit.indexOf('**Audit-failure fallback:**')).replace(/\s+/g, ' ');
expect(parent).toContain("Check the task's terminal status");
expect(parent).toContain('Without successful completion and valid LAST-line JSON, use the audit-failure fallback');
expect(parent).toContain('Require exactly the seven declared fields');
expect(parent).toContain('Missing, extra or invalid fields fail');
expect(parent).toContain('nonnegative integer');
expect(parent).toContain('count sum');
expect(parent).toContain('classification sum equals `total_items`');
expect(parent).toContain('a string `summary`');
expect(parent).toContain('audit-failure fallback');
expect(parent).toContain('Valid no-plan/no-actionable-item reports retain zero counts');
expect(parent).toContain('Valid no-plan/no-actionable reports retain zero counts');
});
test('successful plan audits use the declared seven-field contract without an error field', () => {
const audit = fs.readFileSync(path.join(SHIP_DIR, 'sections/plan-completion.md'), 'utf8');
const line = audit.split('\n').find(line => line.startsWith('{"total_items":N,'));
expect(line).toBeDefined();
const contract = JSON.parse(line!.replace(/:N([,}])/g, ':0$1'));
expect(Object.keys(contract).sort()).toEqual(['total_items', 'done', 'changed', 'partial', 'not_done', 'unverifiable', 'summary'].sort());
expect(contract).not.toHaveProperty('error');
expect(audit).toContain('exactly these seven fields on the LAST LINE');
expect(audit).not.toContain('A non-null `error`');
});
test('CONTENT-SHAPE dispatch invokes validator before falling back to UNVERIFIABLE', () => {
@@ -69,7 +166,7 @@ describe('ship/SKILL.md — Plan Completion gate invariants (VAS-449 remediation
expect(todos).toMatch(/Step 8[^\n]+P1[^\n]+plan/);
expect(todos).toMatch(/Step 5[^\n]+P0[^\n]+deduplicate/);
expect(todos.indexOf('Add approved deferrals')).toBeLessThan(todos.indexOf('Detect completed TODOs'));
expect(todos).toMatch(/unpersisted[^\n]+Step 19/);
expect(todos.replace(/\s+/g, ' ')).toContain("If creation was declined or a write failed, warn and retain unsaved follow-ups in Step 19's PR summary");
});
test('CHANGELOG uses the normal workflow without checkpoint context or squash prerequisites', () => {
@@ -81,15 +178,24 @@ describe('ship/SKILL.md — Plan Completion gate invariants (VAS-449 remediation
test('live evidence recovery distinguishes bookkeeping failure from stale inputs', () => {
const entry = fs.readFileSync(path.join(SHIP_DIR, 'SKILL.md'), 'utf8');
const gate = entry.slice(entry.indexOf('## Step 16:'), entry.indexOf('## Step 17:'));
expect(gate).toContain('Content, command or age mismatch, or no passing live evidence');
expect(gate).toContain('Ledger read/write failure only');
expect(gate).toContain('unchanged final content');
expect(gate).toMatch(/exact command and permitted age, cite its exit,\s+timestamp and log/);
expect(gate).toMatch(/never\s+ledger FRESH/);
expect(gate).toMatch(/Do not rerun green suites solely because the ledger cannot save\s+or read its record/);
expect(gate).toContain('required live RUN must pass');
expect(gate).toMatch(/TODO edits and generated tests are content\s+changes, not ledger-only bookkeeping/);
expect(gate).toContain('If unchanged content cannot be confirmed, STOP');
const text = gate.replace(/\s+/g, ' ');
expect(text).toContain('STALE/MISSING: changed content, command or age, or no proven run');
expect(text).toContain('Only receipt storage/readback failed');
expect(text).toContain('Independently prove unchanged final content, the same command and valid age');
expect(text).toContain('from the successful run\'s evidence');
expect(text).toContain('exact command, exit, timestamp and log');
expect(text).toContain('as **ledger unavailable**');
expect(text).toContain('**ledger unavailable**, never FRESH');
const storage = text.slice(text.indexOf('| Only receipt storage/readback failed |')).split('|')[2];
expect(storage).not.toContain('gstack-evidence run');
expect(storage).toContain('from the successful run\'s evidence');
expect(text).toContain("**New, changed or unwaived test failure:** STOP publication. Run Steps 5–15, starting with Step 5's triage, then return to Step 16 stage 1");
expect(text).toContain('Reuse waivers only for the same verified pre-existing failures and approved scope');
expect(text).toContain('This recovery also applies if a failure appears while reporting in stage 5');
expect(text).toContain('Docs, TODO edits, new/generated tests and fixes make evidence STALE');
expect(storage).toContain('Independently prove unchanged final content, the same command and valid age');
expect(storage).toContain('Without that proof, use STALE/MISSING');
expect(text).toContain('Without that proof, use STALE/MISSING');
});
test('ship contract precedes base detection and fresh remote facts precede distribution decisions', () => {
@@ -99,15 +205,15 @@ describe('ship/SKILL.md — Plan Completion gate invariants (VAS-449 remediation
expect(preflight).toContain('git fetch origin <base>');
expect(preflight).toMatch(/fetch fails[^\n]+STOP/);
expect(entry).not.toContain('auto-generate and commit, or flag');
expect(entry).toContain('commit with Step 15');
expect(entry).toContain('Step 15 commits those tests');
});
test('bisectable commits proceed directly to verification without rewriting existing history', () => {
const entry = fs.readFileSync(path.join(SHIP_DIR, 'SKILL.md'), 'utf8');
const commit = entry.slice(entry.indexOf('## Step 15:'), entry.indexOf('## Step 16:'));
expect(commit).toContain('Create small, logical commits for `git bisect`');
expect(commit).toContain('If all changes are already committed, continue to Step 16');
expect(commit).toContain('never create an empty commit');
expect(commit).toContain('Make bisectable commits');
expect(commit).toContain('if already committed, continue to Step 16');
expect(commit).toMatch(/Never create an empty commit/i);
expect(commit).toContain('Each commit must work independently');
expect(commit).not.toMatch(/checkpoint|WIP|squash|git rebase|git reset/);
expect(entry).not.toMatch(/Step 15\.[012]/);
@@ -116,9 +222,12 @@ describe('ship/SKILL.md — Plan Completion gate invariants (VAS-449 remediation
test('a rejected push stops publication and routes changed content back through verification', () => {
const entry = fs.readFileSync(path.join(SHIP_DIR, 'SKILL.md'), 'utf8');
const push = entry.slice(entry.indexOf('## Step 17:'), entry.indexOf('## Step 20:'));
const recovery = push.replace(/\s+/g, ' ');
expect(push).toMatch(/push fails[^\n]+STOP/);
expect(push).toContain('Step 5');
expect(push).toContain('Step 16');
expect(recovery).toContain('**Non-fast-forward push:**');
expect(recovery).toContain('Run Steps 5–16 before returning to Step 17. Never rewrite history');
expect(recovery).toContain('**Authentication, hook or network failure:**');
expect(recovery).toContain('repeat Step 16 even if content is unchanged before returning to Step 17');
expect(push).toMatch(/never force.push/i);
expect(push).toContain('Only a successful push');
});