Files
gstack/test/ship-skip-requeue.test.ts
T
Garry Tan dcaea52800 v1.91.7.0 feat: add functional QA and pre-publication docs checks (#2983)
* feat: add surface-aware exploratory QA and ship documentation gates

* test: preserve delegated QA setup authority after main integration

* fix(qa): clarify exploration order and preserve report artifacts

* test(qa): follow the shared setup reference directly

* refactor(ship): make verification and recovery routes explicit

* test(ship): align evidence and review guards with explicit routes

* fix(workflows): clarify ship recovery and functional QA evidence

* fix(workflows): clarify approval recovery and full QA coverage

* refactor(workflows): order review transactions and clarify ship state

* fix(ship): clarify final verification and fail closed at publication

* fix(evals): attribute native atomic documentation writes

* fix(ship): clarify recovery and documentation lifecycle guidance

* fix(test): preserve observed native placeholder styling in CI

* fix(codex): report watchdog timeouts without a process-exit race

* Checkpoint functional QA implementation and workflow validation repairs

* Fix documentation and shared-review fixture contracts

* docs: clarify judge reuse and evaluation supervision

* test: align review evidence and selected case contracts

* test: verify append-only documentation checkpoints and recovery

* fix: qualify QA workflows and CI validation repairs

* fix: launch shared-libs fixture scripts on Windows

* fix: qualify QA deadlines, fixture isolation, and shard cleanup

* fix: preserve qualified QA and cancellation repairs

* fix: enforce functional fixture authority and share strict event decoding

* fix: retain free-test evidence and explain recovery

* fix: reject malformed native evidence after decoder consolidation

* test: use reliable capture for telemetry privacy filters

* test: refresh measured quick coverage and document validation costs

* Fix native fixture receipts and preserve VM validation evidence

* Align negative judge controls with upstream clarity policy

* Fix report-only QA preparation and public evidence handling

* Clarify QA-only preparation and current-report preservation

* Stream Ship quality judgments with an explicit 64k response contract

* Validate compact judge reasoning locally with supported wire schema

* Align functional QA fixture instructions with evidence acceptance

* Bind native browser diagnostics to execution evidence and align review verdicts

* Preserve native diagnostic line boundaries

* Serialize functional QA evidence from native captures

* Keep large QA evidence fixture payload out of Windows argv
2026-09-29 06:07:35 -07:00

100 lines
6.6 KiB
TypeScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
import { describe, expect, test } from 'bun:test';
import { readFileSync } from 'node:fs';
import { ALL_HOST_CONFIGS } from '../hosts';
import { generateAdversarialStep, generateCrossReviewDedup, generateSharedCodeReuse } from '../scripts/resolvers/review';
import { HOST_PATHS } from '../scripts/resolvers/types';
const compact = (text: string) => text.replace(/\s+/g, ' ');
const review = compact(readFileSync(new URL('../ship/sections/review-army.md.tmpl', import.meta.url), 'utf8'));
describe.each(ALL_HOST_CONFIGS.map(({ name }) => name))('%s ship skip/requeue contract', host => {
const ctx = { host, skillName: 'ship', tmplPath: '', paths: HOST_PATHS[host] };
const dedup = compact(generateCrossReviewDedup(ctx));
const adversarial = compact(generateAdversarialStep(ctx));
const finish = adversarial.slice(adversarial.indexOf('### Finish the adversarial phase'));
test('unchanged explicit skips are matched before the actionable queue drives a repeat', () => {
const match = finish.indexOf("Apply Step 9.3's matching procedure");
expect(match).toBeGreaterThan(-1);
expect(match).toBeLessThan(finish.indexOf('2. **Fixes queued'));
expect(finish).toContain('Only unmatched or reopened findings remain queued');
expect(dedup).toContain('Only explicit `skipped` actions qualify');
expect(dedup).toContain('If both history and the invocation action list lack decisions, classify normally');
expect(dedup).toContain('Revalidated Skips suppress repeat questions and fixes');
expect(dedup).toContain('Report the suppressed count once if nonzero');
expect(dedup).toContain('reopen the finding; unrelated edits do not');
const steps = ['1. **Validate severity.**', '2. **Read decisions.**',
'3. **Match evidence.**', '4. **Match shared-code structurally.**', '5. **Apply dispositions.**'];
const positions = steps.map(step => dedup.indexOf(step));
expect(positions.every(position => position >= 0)).toBe(true);
expect(positions).toEqual([...positions].sort((a, b) => a - b));
expect(review).toContain('Classify only unmatched or reopened findings as AUTO-FIX or ASK');
expect(review).toContain('after Step 9.3 matches all sources, including queued Steps 10–11 findings');
});
test('dedup and persistence include queued sources and preserve the explicit decision', () => {
expect(dedup).toContain('exploratory QA and queued Steps 10–11 findings');
expect(review).toContain('checklist, specialist, exploratory QA and queued Steps 10–11 records');
expect(review).toContain('Save each explicit Skip immediately in the invocation action list');
expect(review).toContain('preserve `advisory`, `evidence_paths` and `helper_target`');
});
test('working-tree changes and new evidence reopen matching identities', () => {
expect(dedup).toContain('git diff --name-only <prior-review-commit>');
expect(dedup).not.toContain('git diff --name-only <prior-review-commit> HEAD');
expect(dedup).toContain('committed, staged, unstaged and non-ignored untracked source');
expect(dedup).toContain('Changed inputs, proposal, behavior, risk or new evidence reopen the finding');
expect(dedup).toContain('honoring later user decisions');
expect(dedup).toContain('same fingerprint, advisory/defect kind and scope');
expect(dedup).toContain('Compare supporting source and finding evidence with the saved decision');
expect(dedup).toContain('as a shortlist, not proof');
});
test('fixed regressions and missing Skip proof are not suppressed', () => {
expect(dedup).toContain('never `fixed`, `auto-fixed` or unanswered questions');
expect(dedup).toContain('Missing proof or unknown comparisons require a fresh decision, not suppression');
expect(finish).toContain('Keep scoped approvals');
expect(dedup).toContain('Require the same fingerprint, advisory/defect kind and scope');
expect(finish).toContain('Unvalidated historical Skips stay unmatched for the full Step 9 repeat below');
expect(finish).toContain('never jump to 9.3 or mint a late REVIEW_START');
});
test('shared-code and advisory collisions retain stricter identity checks', () => {
expect(dedup).toContain('they cannot suppress defects');
expect(dedup).toContain('remove `advisory`, never downgrade severity');
expect(dedup).toContain('Reject contradictory saved decisions');
expect(dedup).toContain('requires re-reading all callers (including indirect callers) and the helper destination');
expect(dedup).toContain('Missing metadata never permits ordinary line matching');
expect(dedup).toContain('Prior-review reuse additionally requires the checker below; invocation decisions cannot replace it');
const checker = compact(generateSharedCodeReuse(ctx));
expect(checker).toContain('Only `reusable: true` permits suppression');
expect(checker).toContain('False, command failure or unreadable output requires fresh source review');
expect(checker).toContain('Do not supply your own snapshot, prior record or coverage');
});
test('skipped defects stay unresolved and required failures stay failed', () => {
expect(dedup).toContain('not unresolved defects: retain them in counts, status and the final report');
expect(dedup).toContain('Keep required-probe failures failed');
expect(review).toContain('Skipping a fix is not risk acceptance or a passing probe');
expect(review).toContain('Failed, blocked, inconclusive or not-run required probes mean false, never clean');
expect(review).toContain('VERIFY_RESULT stays fail');
expect(adversarial).toContain('retain the acknowledged findings and failed gate; do not report a clean review');
});
test('fresh review after edits, native coverage and cycle limits remain required', () => {
expect(finish).toContain('**Required native review incomplete:** STOP');
expect(finish).toContain('Insert Steps 9, 10 and 11 before the pending Step 11.5');
expect(finish).toContain("never resets Step 9's three-cycle fix limit");
expect(review).toContain('Set CYCLES to 0 on first entry only');
expect(review).toContain('Increment CYCLES once if fixes were applied');
expect(review).toContain('do not run a fourth fixing cycle');
expect(finish).toContain('then continue to Step 11.5. Never jump directly to release preparation');
});
});
test('ship invocation matching does not change standalone review routing', () => {
const ctx = { host: 'claude' as const, skillName: 'review', tmplPath: '', paths: HOST_PATHS.claude };
expect(generateCrossReviewDedup(ctx)).not.toContain('5. **Apply dispositions.**');
expect(generateAdversarialStep(ctx)).not.toContain('### Finish the adversarial phase');
});