Files
gstack/test/ship-reentry-gates.test.ts
T
Garry Tan dcaea52800 v1.91.7.0 feat: add functional QA and pre-publication docs checks (#2983)
* feat: add surface-aware exploratory QA and ship documentation gates

* test: preserve delegated QA setup authority after main integration

* fix(qa): clarify exploration order and preserve report artifacts

* test(qa): follow the shared setup reference directly

* refactor(ship): make verification and recovery routes explicit

* test(ship): align evidence and review guards with explicit routes

* fix(workflows): clarify ship recovery and functional QA evidence

* fix(workflows): clarify approval recovery and full QA coverage

* refactor(workflows): order review transactions and clarify ship state

* fix(ship): clarify final verification and fail closed at publication

* fix(evals): attribute native atomic documentation writes

* fix(ship): clarify recovery and documentation lifecycle guidance

* fix(test): preserve observed native placeholder styling in CI

* fix(codex): report watchdog timeouts without a process-exit race

* Checkpoint functional QA implementation and workflow validation repairs

* Fix documentation and shared-review fixture contracts

* docs: clarify judge reuse and evaluation supervision

* test: align review evidence and selected case contracts

* test: verify append-only documentation checkpoints and recovery

* fix: qualify QA workflows and CI validation repairs

* fix: launch shared-libs fixture scripts on Windows

* fix: qualify QA deadlines, fixture isolation, and shard cleanup

* fix: preserve qualified QA and cancellation repairs

* fix: enforce functional fixture authority and share strict event decoding

* fix: retain free-test evidence and explain recovery

* fix: reject malformed native evidence after decoder consolidation

* test: use reliable capture for telemetry privacy filters

* test: refresh measured quick coverage and document validation costs

* Fix native fixture receipts and preserve VM validation evidence

* Align negative judge controls with upstream clarity policy

* Fix report-only QA preparation and public evidence handling

* Clarify QA-only preparation and current-report preservation

* Stream Ship quality judgments with an explicit 64k response contract

* Validate compact judge reasoning locally with supported wire schema

* Align functional QA fixture instructions with evidence acceptance

* Bind native browser diagnostics to execution evidence and align review verdicts

* Preserve native diagnostic line boundaries

* Serialize functional QA evidence from native captures

* Keep large QA evidence fixture payload out of Windows argv
2026-09-29 06:07:35 -07:00

77 lines
5.0 KiB
TypeScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
import { expect, test } from 'bun:test';
import { readFileSync } from 'node:fs';
import { generatePlanCompletionGateShip } from '../scripts/resolvers/review';
import { generateTestBootstrap } from '../scripts/resolvers/testing';
import { HOST_PATHS } from '../scripts/resolvers/types';
const read = (file: string) => readFileSync(new URL(`../${file}`, import.meta.url), 'utf8');
const compact = (text: string) => text.replace(/\s+/g, ' ');
const ctx = { host: 'claude' as const, skillName: 'ship', tmplPath: '', paths: HOST_PATHS.claude };
test('ship STOP blocks advancement while retaining the stated repair route', () => {
const entry = compact(read('ship/SKILL.md.tmpl'));
expect(entry).toContain('STOP blocks advancement until the stated repair/resume route clears; without one, end this attempt');
expect(entry).toContain('Answer each AskUserQuestion before continuing');
const review = compact(read('ship/sections/review-army.md.tmpl'));
expect(review).toContain('**Dispatched reviewer output missing:** STOP');
expect(review).toContain('Retain queued fixes and restore coverage');
expect(review).toContain('**Third fixing cycle reached (`CYCLES >= 3`):** STOP');
expect(entry).toContain('Routine authorization never waives those gates or their required user decisions');
});
test('a new ship bootstrap choice overrides only the saved decline, not framework selection', () => {
const ship = generateTestBootstrap(ctx);
const decline = ship.slice(ship.indexOf('**If BOOTSTRAP_DECLINED**'), ship.indexOf('**If NO ecosystem marker matched:**'));
expect(compact(decline)).toContain("Step 5's explicit Add tests choice overrides that marker for this invocation only");
expect(compact(decline)).toContain('continue to runtime detection and B2–B3, including framework approval');
expect(decline).not.toContain('rm ');
expect(ship.indexOf('**If ANY existing-test evidence appears**')).toBeLessThan(ship.indexOf('**If BOOTSTRAP_DECLINED**'));
const qa = generateTestBootstrap({ ...ctx, skillName: 'qa' });
expect(qa).toContain('**If BOOTSTRAP_DECLINED** appears: Print "Test bootstrap previously declined — skipping." **Skip the rest of bootstrap.**');
expect(qa).not.toContain("Step 5's explicit Add tests choice");
});
test('an unverified item answered not done enters the existing decision before continuing', () => {
const gate = generatePlanCompletionGateShip(ctx);
const exits = compact(gate.slice(gate.indexOf('**Exit conditions:**'), gate.indexOf('**Cap.**')));
expect(exits).toContain('Any N: STOP and report that item as NOT DONE');
expect(exits).toContain('Resume only after its required work is verified');
expect(exits).toContain('no second deferral choice');
expect(exits).not.toContain('re-running /ship');
expect(gate).toContain('Per-item confirmation is mandatory');
expect(gate).toContain('with the user\'s free-text evidence');
});
test('docs reentry distinguishes the initial audit from same-invocation accepted evidence', () => {
const docs = compact(read('ship/sections/documentation.md.tmpl'));
const entry = docs.slice(0, docs.indexOf('## Prepare the candidate'));
expect(entry).toContain('First entry always launches the initial audit');
expect(entry).toContain('On reentry, reuse only this invocation\'s validated audit or named-risk decision');
expect(entry).toContain('Reentry never resets the count or authorizes a launch');
expect(entry).toContain("this invocation's validated audit or named-risk decision");
expect(entry).toContain('base/input hashes still match');
expect(entry).toContain('retain its actual status and scope');
expect(entry).toContain('Otherwise use Blocked recovery, not an unconditional launch');
expect(docs).toContain('never a third attempt, even after Step 16 changes');
expect(docs).toContain('never reuse an audit across invocations');
expect(docs).toContain('Unconfirmed writers, ownership violations, unauthorized Git mutation and redaction/security gates cannot be waived');
});
test('late behavioral repairs rebuild before the docs decision and commit only remaining changes', () => {
const entry = compact(read('ship/SKILL.md.tmpl'));
expect(entry).toContain('A range ending at Step 14 does not enter Step 14.5');
expect(entry).toContain('rebuild and compare again before stage 3 decides documentation freshness');
const commit = entry.slice(entry.indexOf('### 5. Report, then push'), entry.indexOf('## Step 17:'));
expect(commit).toContain('left uncommitted after Step 15');
expect(commit).toContain('never create an empty commit');
expect(commit).toContain('Preserve unrelated user files');
});
test('the title is prefixed once before the exact scanned value is published', () => {
const body = compact(read('ship/sections/pr-body.md.tmpl'));
expect(body).toContain("Use Step 18's `NEW_TITLE` unchanged; its version prefix is already present");
expect(body).not.toContain('`NEW_TITLE`, prefixed with');
expect(body).toContain('--title "$NEW_TITLE"');
expect(body).toContain('the same scanned `NEW_TITLE`');
});