v1.91.7.0 feat: add functional QA and pre-publication docs checks (#2983)

* feat: add surface-aware exploratory QA and ship documentation gates

* test: preserve delegated QA setup authority after main integration

* fix(qa): clarify exploration order and preserve report artifacts

* test(qa): follow the shared setup reference directly

* refactor(ship): make verification and recovery routes explicit

* test(ship): align evidence and review guards with explicit routes

* fix(workflows): clarify ship recovery and functional QA evidence

* fix(workflows): clarify approval recovery and full QA coverage

* refactor(workflows): order review transactions and clarify ship state

* fix(ship): clarify final verification and fail closed at publication

* fix(evals): attribute native atomic documentation writes

* fix(ship): clarify recovery and documentation lifecycle guidance

* fix(test): preserve observed native placeholder styling in CI

* fix(codex): report watchdog timeouts without a process-exit race

* Checkpoint functional QA implementation and workflow validation repairs

* Fix documentation and shared-review fixture contracts

* docs: clarify judge reuse and evaluation supervision

* test: align review evidence and selected case contracts

* test: verify append-only documentation checkpoints and recovery

* fix: qualify QA workflows and CI validation repairs

* fix: launch shared-libs fixture scripts on Windows

* fix: qualify QA deadlines, fixture isolation, and shard cleanup

* fix: preserve qualified QA and cancellation repairs

* fix: enforce functional fixture authority and share strict event decoding

* fix: retain free-test evidence and explain recovery

* fix: reject malformed native evidence after decoder consolidation

* test: use reliable capture for telemetry privacy filters

* test: refresh measured quick coverage and document validation costs

* Fix native fixture receipts and preserve VM validation evidence

* Align negative judge controls with upstream clarity policy

* Fix report-only QA preparation and public evidence handling

* Clarify QA-only preparation and current-report preservation

* Stream Ship quality judgments with an explicit 64k response contract

* Validate compact judge reasoning locally with supported wire schema

* Align functional QA fixture instructions with evidence acceptance

* Bind native browser diagnostics to execution evidence and align review verdicts

* Preserve native diagnostic line boundaries

* Serialize functional QA evidence from native captures

* Keep large QA evidence fixture payload out of Windows argv
This commit is contained in:
Garry Tan authored and GitHub committed 2026-09-29 06:07:35 -07:00
1 parent 65bfb0ce49
commit dcaea52800
333 files changed
+41755 -7357

No files matched your search

+36 -15
View File
@@ -63,7 +63,7 @@ describe('workflow judge excerpts', () => {
test('expands ship sections in execution order, not alphabetical order', () => {
const text = readWorkflowExcerpt('ship/SKILL.md', '# Ship:', '## Important Rules');
const headings = ['## Step 3:', '## Step 4:', '## Step 7:', '## Step 8:', '## Step 9:', '## Step 10:', '## Step 11:', '## Step 12:', '## Step 13:', '## Step 14:'];
const headings = ['## Step 3:', '## Step 4:', '## Step 7:', '## Step 8:', '## Step 9:', '## Step 10:', '## Step 11:', '## Step 11.5:', '## Step 12:', '## Step 13:', '## Step 14:'];
const indices = headings.map(heading => text.indexOf(heading));
expect(indices.every(index => index >= 0)).toBe(true);
expect(indices).toEqual([...indices].sort((a, b) => a - b));
@@ -88,12 +88,29 @@ describe('workflow judge excerpts', () => {
test('ship review shortcuts retain dedup and fixes repeat the whole review cycle', () => {
const text = readWorkflowExcerpt('ship/SKILL.md', '# Ship:', '## Important Rules');
expect(text).toContain('Continue to Step 9.3 (cross-review dedup)');
expect(text).toContain("Continue to Step 9.2 with the core/design-lite findings and an empty specialist list, then the parent's Exploratory QA step and Step 9.3 (cross-review dedup)");
expect(text).toContain('## Step 9.4: Fix-First and persistence');
expect(text).toContain('including design, specialists, Red Team, and dedup');
expect(text.replace(/\s+/g, ' ')).toContain('Run checklist/design, specialists (9.1), merge/Red Team (9.2), exploratory QA (9.2.1), dedup (9.3), then fixes and logging (9.4)');
expect(text.replace(/\s+/g, ' ')).toContain('**Fixes applied below the cap:** Insert Step 5, affected Steps 6–8 and all of Step 9 before the pending Step 10 in the work list. Tests must pass or retain approval for the same verified pre-existing failures and scope');
const audit = text.slice(text.indexOf('## Step 7:'), text.indexOf('## Step 8:'));
expect(audit).not.toContain('Scope Challenge');
expect(text).toContain('Ship anyway retains VERIFY_RESULT=fail');
expect(text.replace(/\s+/g, ' ')).toContain('Keep actual outcomes and incomplete flags; VERIFY_RESULT stays fail for plan-check exceptions');
});
test('ship excerpt preserves readable detours, audit fallback and final input decisions', () => {
const text = readWorkflowExcerpt('ship/SKILL.md', '# Ship:', '## Important Rules').replace(/\s+/g, ' ');
expect(text).toContain('For another repair, repeat rule 2 without discarding pending work');
expect(text).toContain('A further Step 9 fix affecting 6–8 makes the list `5 → 6 → 7 → 8 → 9 → 10 → 11 → 11.5`');
expect(text).toContain('The unchanged release steps follow');
expect(text).toContain('All three snapshots must match');
expect(text).toContain('does not mean the failed or unrun probes passed');
expect(text).toContain('Fallback recovers the audit; it does not pass or bypass the coverage gate');
expect(text).toContain('Skip only the plan completion audit');
expect(text).toContain('Continue with Step 8.1, Scope Drift and Prior Learnings');
expect(text).toContain('Step 9 QA still runs');
expect(text).toContain('Use this example only after confirming that every allowed edit is release metadata');
expect(text).not.toContain('Every listed change below is metadata:');
expect(text).not.toContain('No plan file found:** Skip entirely');
});
test('a sliced section is not appended again with its generated header', () => {
@@ -117,7 +134,12 @@ describe('workflow judge excerpts', () => {
expect(text).toContain('never create an empty commit');
const review = text.slice(text.indexOf('## Step 9:'), text.indexOf('## Step 10:'));
expect(review.indexOf('## Confidence Calibration')).toBeLessThan(review.indexOf('1. Read'));
expect(review).toContain('Continue to Step 10 only after a completed, converged review is persisted');
const flat = review.replace(/\s+/g, ' ');
expect(flat).toContain('**No edits in this pass:** Resolve the required-probe gate below. Only after it clears may you continue to Step 10');
expect(flat).toContain('**Dispatched reviewer output missing:** STOP');
expect(flat).toContain('Retain queued fixes and restore coverage');
expect(flat).toContain('**Third fixing cycle reached (`CYCLES >= 3`):** STOP and report recurring findings with `converged:false`; do not run a fourth fixing cycle');
expect(flat).toContain('With completed checklist and dispatched reviewers, failed/unavailable required probes block continuation');
});
test('ship approval gates stay outside the subagent prompts', () => {
@@ -131,7 +153,7 @@ describe('workflow judge excerpts', () => {
expect(section.indexOf(gate)).toBeGreaterThan(section.indexOf('\n````\n'));
}
expect(text).toContain('"partial":N,"not_done":N');
expect(text).toContain('each Y response\'s evidence and each D response\'s dropped item');
expect(text).toContain('each Y\'d item with the user\'s free-text evidence and each D\'d item with "intentionally dropped"');
});
test('expands a body before the end marker in the skeleton', () => {
@@ -163,7 +185,7 @@ describe('workflow judge excerpts', () => {
const { skillPath, startMarker, endMarker } = ENG_REVIEW_EXCERPT;
const eng = readWorkflowExcerpt(skillPath, startMarker, endMarker);
const stages = ['## Review preparation', '## Retrospective learning', '## Confidence Calibration', '## Decision procedure',
'### 1. Establish current state', '## Scope Challenge', '### A. Assess the target',
'### Prepare an unanswered choice', '## Scope Challenge', '### A. Assess the target',
'### B. Resolve complexity selectors', '### C. Resolve findings', '## Review Sections',
'### 1. Architecture review', '### 2. Code quality review', '### 3. Test review', '### 4. Performance review']
.map(heading => eng.indexOf(heading));
@@ -172,11 +194,10 @@ describe('workflow judge excerpts', () => {
expect(eng.match(/^## Decision procedure$/gm)).toHaveLength(1);
const procedure = eng.slice(eng.indexOf('## Decision procedure'), eng.indexOf('## Scope Challenge'));
const headings = marked.lexer(procedure).filter(token => token.type === 'heading' && token.depth === 3);
expect(headings.map(token => token.text)).toEqual(['1. Establish current state', '2. Separate independent choices', '3. Compare one choice',
'4. Save the pending record', '5. Ask and wait', '6. Apply and refresh']);
expect(procedure).toContain("### 4. Save the pending record");
expect(procedure).toContain('### 5. Ask and wait');
expect(procedure).toContain("### 6. Apply and refresh");
expect(headings.map(token => token.text)).toEqual(['Prepare an unanswered choice', 'Send once and wait', 'Record the answer']);
expect(procedure).toContain("**Pending-record checkpoint.**");
expect(procedure).toContain('### Send once and wait');
expect(procedure).toContain("### Record the answer");
const outputs = ['### TODOS.md updates', '## Approval readiness', '## Required outputs', '## Implementation Tasks',
'### Unresolved decisions', '### Completion summary', '## Plan File Review Report',
'### Write to the report file', '## Review Log'].map(heading => eng.indexOf(heading));
@@ -276,7 +297,7 @@ console.log(JSON.stringify({calls, results}));
expectOutsideReviewControlFlow(eng, '**Construct the plan review prompt**');
expect(eng).toContain('Agreement between reviewers is evidence, not approval');
expect(eng).toContain('new or reopened choices still need their own answers');
const pendingDecision = eng.slice(eng.indexOf('### 5. Ask and wait'), eng.indexOf("### 6. Apply and refresh"));
const pendingDecision = eng.slice(eng.indexOf('### Send once and wait'), eng.indexOf("### Record the answer"));
expect(pendingDecision).toContain("**STOP until the actual answer arrives.**");
expect(pendingDecision.replace(/\s+/g, ' ')).toContain("Do not apply a remedy, make another call, start the next section or call ExitPlanMode while the choice awaits an answer");
expect(eng.replace(/\s+/g, ' ')).toContain("Use a scoped Edit to save this record and only the authorized working-plan amendments. Leave other choices unchanged");
@@ -330,8 +351,8 @@ console.log(JSON.stringify({calls, results}));
test('ship commits logical chunks without rewriting existing checkpoint commits', () => {
const text = readWorkflowExcerpt('ship/SKILL.md', '# Ship:', '## Important Rules');
const commit = text.slice(text.indexOf('## Step 15:'), text.indexOf('## Step 16:'));
expect(commit).toContain('Create small, logical commits');
expect(commit).toContain('If all changes are already committed, continue to Step 16');
expect(commit).toContain('Make bisectable commits');
expect(commit).toContain('if already committed, continue to Step 16');
expect(commit).toContain('Each commit must work independently');
expect(commit).not.toMatch(/rebase|reset|squash|fixup|WIP_TODO|gstack-context/);
expect(text).not.toContain('Step 15.0');