v1.91.7.0 feat: add functional QA and pre-publication docs checks (#2983)

* feat: add surface-aware exploratory QA and ship documentation gates

* test: preserve delegated QA setup authority after main integration

* fix(qa): clarify exploration order and preserve report artifacts

* test(qa): follow the shared setup reference directly

* refactor(ship): make verification and recovery routes explicit

* test(ship): align evidence and review guards with explicit routes

* fix(workflows): clarify ship recovery and functional QA evidence

* fix(workflows): clarify approval recovery and full QA coverage

* refactor(workflows): order review transactions and clarify ship state

* fix(ship): clarify final verification and fail closed at publication

* fix(evals): attribute native atomic documentation writes

* fix(ship): clarify recovery and documentation lifecycle guidance

* fix(test): preserve observed native placeholder styling in CI

* fix(codex): report watchdog timeouts without a process-exit race

* Checkpoint functional QA implementation and workflow validation repairs

* Fix documentation and shared-review fixture contracts

* docs: clarify judge reuse and evaluation supervision

* test: align review evidence and selected case contracts

* test: verify append-only documentation checkpoints and recovery

* fix: qualify QA workflows and CI validation repairs

* fix: launch shared-libs fixture scripts on Windows

* fix: qualify QA deadlines, fixture isolation, and shard cleanup

* fix: preserve qualified QA and cancellation repairs

* fix: enforce functional fixture authority and share strict event decoding

* fix: retain free-test evidence and explain recovery

* fix: reject malformed native evidence after decoder consolidation

* test: use reliable capture for telemetry privacy filters

* test: refresh measured quick coverage and document validation costs

* Fix native fixture receipts and preserve VM validation evidence

* Align negative judge controls with upstream clarity policy

* Fix report-only QA preparation and public evidence handling

* Clarify QA-only preparation and current-report preservation

* Stream Ship quality judgments with an explicit 64k response contract

* Validate compact judge reasoning locally with supported wire schema

* Align functional QA fixture instructions with evidence acceptance

* Bind native browser diagnostics to execution evidence and align review verdicts

* Preserve native diagnostic line boundaries

* Serialize functional QA evidence from native captures

* Keep large QA evidence fixture payload out of Windows argv
This commit is contained in:
Garry Tan authored and GitHub committed 2026-09-29 06:07:35 -07:00
1 parent 65bfb0ce49
commit dcaea52800
333 files changed
+41755 -7357

No files matched your search

+113 -47
View File
@@ -359,7 +359,10 @@ describe('gen-skill-docs', () => {
// Aside is the primary browser: every browsing skill renders the Aside
// contract ({{ASIDE_SETUP}}); the browse binary is its fallback.
const qaTmpl = fs.readFileSync(path.join(ROOT, 'qa', 'SKILL.md.tmpl'), 'utf-8');
expect(qaTmpl).toContain('{{ASIDE_SETUP}}');
expect(qaTmpl).not.toContain('{{ASIDE_SETUP}}');
expect(qaTmpl).toContain('{{SECTION:browser-setup}}');
expect(fs.readFileSync(path.join(ROOT, 'qa/sections/browser-setup.md.tmpl'), 'utf8'))
.toContain('{{ASIDE_SETUP}}');
expect(browseTmpl).toContain('{{ASIDE_SETUP}}');
});
@@ -572,22 +575,55 @@ describe('gen-skill-docs', () => {
}
});
test('qa and qa-only templates use QA_METHODOLOGY placeholder', () => {
// qa carve: the macro moved into the section template (the skeleton
// carries the STOP-Read pointer); qa-only remains an inline monolith.
test('qa and qa-only load the shared QA_METHODOLOGY through exploration', () => {
const qaSkeletonTmpl = fs.readFileSync(path.join(ROOT, 'qa', 'SKILL.md.tmpl'), 'utf-8');
expect(qaSkeletonTmpl).toContain('{{SECTION:qa-patterns}}');
expect(qaSkeletonTmpl).toContain('{{SECTION:exploratory}}');
expect(qaSkeletonTmpl).not.toContain('{{QA_METHOD_READS}}');
expect(qaSkeletonTmpl).toContain("Follow the shared section's ordered preparation");
expect(qaSkeletonTmpl).not.toContain('{{QA_METHODOLOGY}}');
const qaSectionTmpl = fs.readFileSync(path.join(ROOT, 'qa', 'sections', 'qa-patterns.md.tmpl'), 'utf-8');
expect(qaSectionTmpl).toContain('{{QA_METHODOLOGY}}');
const qaOnlyTmpl = fs.readFileSync(path.join(ROOT, 'qa-only', 'SKILL.md.tmpl'), 'utf-8');
expect(qaOnlyTmpl).toContain('{{QA_METHODOLOGY}}');
expect(qaOnlyTmpl).not.toContain('{{QA_METHODOLOGY}}');
expect(qaOnlyTmpl).toContain('{{SECTION:exploratory}}');
expect(qaOnlyTmpl).not.toContain('{{QA_METHOD_READS}}');
expect(qaOnlyTmpl).toContain('Load the shared preparation gate now');
expect(qaOnlyTmpl).toContain('Use the shared section already loaded above');
for (const skill of ['qa', 'qa-only']) {
expect(fs.readFileSync(path.join(ROOT, skill, 'sections/exploratory.md.tmpl'), 'utf8'))
.toContain('{{QA_EXPLORATORY}}');
const entry = fs.readFileSync(path.join(ROOT, skill, 'SKILL.md'), 'utf8');
const explorer = fs.readFileSync(path.join(ROOT, skill, 'sections/exploratory.md'), 'utf8');
const directoryRead = skill === 'qa'
? 'Read `sections/scope.md` relative to the installed `qa`/`gstack-qa` SKILL.md directory'
: "Use this host's installed `qa`/`gstack-qa` SKILL.md directory for these reads";
expect(entry).toContain('sections/exploratory.md');
expect(explorer).toContain(directoryRead);
for (const method of ['system-functional', 'qa-patterns']) {
const methodRead = `Read \`sections/${method}.md\` in full.`;
expect(entry).not.toContain(methodRead);
expect(explorer).toContain(methodRead);
expect(explorer.split(methodRead)).toHaveLength(2);
expect(explorer.indexOf(directoryRead)).toBeLessThan(explorer.indexOf(methodRead));
const directory = path.resolve(ROOT, skill, skill === 'qa-only' ? '../qa' : '.');
expect(fs.realpathSync(path.join(directory, 'sections', `${method}.md`)))
.toBe(path.join(ROOT, 'qa', 'sections', `${method}.md`));
}
expect(explorer).toContain('**Browser surfaces only:**\nRead `sections/qa-patterns.md` in full.');
expect(explorer).toContain('Complete these Reads in order before writing charters or probing');
expect(explorer).toContain('Do not repeat a Read already completed in this invocation');
expect(explorer.indexOf('Read `sections/qa-patterns.md`')).toBeLessThan(explorer.indexOf('Write a **charter**'));
}
});
test('QA_METHODOLOGY appears expanded in both qa and qa-only generated files', () => {
test('QA_METHODOLOGY is expanded in the shared resource referenced by qa and qa-only', () => {
const qaContent = readSkillUnion('qa'); // carved: methodology lives in qa/sections/qa-patterns.md
const qaOnlyContent = fs.readFileSync(path.join(ROOT, 'qa-only', 'SKILL.md'), 'utf-8');
const qaOnlyUnion = readSkillUnion('qa-only');
expect(qaOnlyUnion).toContain("Use this host's installed `qa`/`gstack-qa` SKILL.md directory for these reads");
expect(qaOnlyUnion).toContain('**Browser surfaces only:**\nRead `sections/qa-patterns.md` in full.');
expect(qaOnlyUnion).not.toContain('Health Score Rubric');
const qaOnlyContent = qaOnlyUnion + fs.readFileSync(path.join(ROOT, 'qa/sections/qa-patterns.md'), 'utf8');
// Both should contain the health score rubric
expect(qaContent).toContain('Health Score Rubric');
@@ -803,16 +839,16 @@ describe('REVIEW_DASHBOARD resolver', () => {
test('dashboard treats review as a valid Eng Review source', () => {
const content = readShipUnion();
expect(content).toContain('plan-eng-review, review, plan-design-review');
expect(content).toContain('`review` (diff-scoped pre-landing review)');
expect(content).toContain('`plan-eng-review` (plan-stage architecture review)');
expect(content).toContain('from either \\`review\\` or \\`plan-eng-review\\`');
expect(content).toContain('| Eng Review | `review` or `plan-eng-review` | (DIFF) or (PLAN) |');
expect(content).toContain('**Content-first rule:** For `review`');
expect(content).toContain('**Plan records** (plan-ceo-review, plan-eng-review');
expect(content.replace(/\s+/g, ' ')).toContain('CLEARED requires the selected Eng Review to be `clean`, within 7 days and fresh under step 2');
});
test('shared dashboard propagates review source to plan-eng-review', () => {
const content = readSkillUnion('plan-eng-review'); // carved: review body moved to section
expect(content).toContain('plan-eng-review, review, plan-design-review');
expect(content).toContain('`review` (diff-scoped pre-landing review)');
expect(content).toContain('| Eng Review | `review` or `plan-eng-review` | (DIFF) or (PLAN) |');
expect(content).toContain('**Content-first rule:** For `review`');
});
test('resolver output contains key dashboard elements', () => {
@@ -833,7 +869,7 @@ describe('REVIEW_DASHBOARD resolver', () => {
test('dashboard includes staleness detection prose', () => {
const content = readSkillUnion('plan-ceo-review'); // carved: dashboard moved to section
expect(content).toContain('Staleness detection');
expect(content).toContain('**2. Check freshness before choosing a verdict.**');
expect(content).toContain('commit');
});
@@ -1075,7 +1111,8 @@ describe('TEST_COVERAGE_AUDIT placeholders', () => {
'utf-8',
);
expect(reviewArmySection).toContain('"advisory": true');
expect(reviewArmySection).toContain('quality score over NON-advisory findings only');
expect(reviewArmySection).toContain('Only specialist findings enter this header and `quality_score`; core findings do not');
expect(reviewArmySection).toContain('Use the merged NON-advisory specialist findings for both counts and score');
expect(reviewArmySection).toContain('Simplification: lean already — nothing to cut.');
expect(reviewArmySection).toContain('net: -N lines possible');
expect(reviewArmySection).toContain('--simplification');
@@ -1136,9 +1173,10 @@ describe('TEST_COVERAGE_AUDIT placeholders', () => {
});
test('ship SKILL.md contains re-run idempotency behavior', () => {
expect(shipSkill).toContain('Re-run behavior (idempotency)');
expect(shipSkill).toContain('Every invocation repeats verification:');
expect(shipSkill).toContain('Prior execution never exempts verification.');
expect(shipSkill).toContain('**Route:**');
expect(shipSkill.replace(/\s+/g, ' ')).toContain('integrate (1–3) → test and review (4–11.5) → prepare the release (12–15) → verify frozen content (16) → push and publish (17–21)');
expect(shipSkill.replace(/\s+/g, ' ')).toContain('Every new invocation repeats Steps 1–16, including both reviews and the docs audit');
expect(shipSkill.replace(/\s+/g, ' ')).toContain('Steps 12, 17 and 19 prevent duplicate bumps, pushes and PRs, never verification');
});
});
@@ -1228,9 +1266,9 @@ describe('PLAN_FILE_REVIEW_REPORT resolver', () => {
for (const output of [confidence, dashboard, report, outside]) expect(output).not.toContain('\\`');
expect(confidence).toBe(generateConfidenceCalibration({...ctx, skillName: 'plan-ceo-review'}).replaceAll('\\`', '`'));
const ceoDashboard = generateReviewDashboard({...ctx, skillName: 'plan-ceo-review'}).replaceAll('\\`', '`');
const ceoVoiceSource = 'From gstack-review-read output, use entries whose skill is `autoplan-voices` or `design-outside-voices` for the coverage detail below the dashboard.';
expect(dashboard).toContain(ceoVoiceSource);
expect(ceoDashboard).toContain(ceoVoiceSource);
const ceoVoiceSource = 'Below the dashboard, group `autoplan-voices` and `design-outside-voices` by workflow run and phase';
expect(dashboard.replace(/\s+/g, ' ')).toContain(ceoVoiceSource);
expect(ceoDashboard.replace(/\s+/g, ' ')).toContain(ceoVoiceSource);
expect(ceoDashboard).toBe(dashboard);
for (const field of ['status', 'unresolved', 'critical_gaps', 'issues_found', 'mode', 'commit']) {
expect(report).toContain('`' + field + '`');
@@ -1331,25 +1369,44 @@ describe('PLAN_VERIFICATION_EXEC placeholder', () => {
expect(shipSkill).toContain('Plan Verification');
});
test('references /qa-only invocation', () => {
expect(shipSkill).toContain('qa-only/SKILL.md');
expect(shipSkill).toContain('qa-only');
test('references the shared explorer without invoking an entire QA workflow', () => {
const resource = "From the installed /ship SKILL.md's directory, Read `../qa/sections/exploratory.md` in full";
expect(shipSkill).toContain(resource);
const load = shipSkill.indexOf(resource);
const preflight = shipSkill.indexOf('Run the shared preflight;');
const probes = shipSkill.indexOf('**3. Run smoke and plan checks.**');
expect(load).toBeGreaterThan(-1);
expect(preflight).toBeGreaterThan(load);
expect(probes).toBeGreaterThan(preflight);
const shared = fs.readFileSync(path.join(ROOT, 'qa/sections/exploratory.md'), 'utf8');
const selection = shared.indexOf('in full and select the surfaces');
const methods = shared.indexOf('Read `sections/system-functional.md`');
expect(selection).toBeGreaterThan(shared.indexOf('Read `sections/scope.md`'));
expect(methods).toBeGreaterThan(selection);
expect(shared.indexOf('Write a **charter**')).toBeGreaterThan(methods);
expect(shipSkill.slice(preflight, probes)).toContain("For browsers, Read QA's `sections/browser-setup.md`");
expect(shipSkill).toContain('Do not invoke an entire QA skill or start probes here');
});
test('contains dev-server discovery (CLAUDE.md first, then a port probe)', () => {
// Fork port wave 2: the hardcoded 4-port list became read-CLAUDE.md-or-
// probe; the probe loops common ports instead of naming each once.
expect(shipSkill).toContain('CLAUDE.md first');
expect(shipSkill).toContain('http://localhost:$_p');
expect(shipSkill).toContain('NO_SERVER');
test('keeps declared browser URLs separate from native functional probes', () => {
expect(shipSkill).toContain('items use the declared project/plan dev URL');
expect(shipSkill).toContain('functional items use native tools without discovering a web server');
expect(shipSkill.replace(/\s+/g, ' ')).toContain('An API URL is not automatically a page');
});
test('skips gracefully when no verification section', () => {
expect(shipSkill).toContain('No verification steps found in plan');
test('retains automatic exploration when there is no plan or verification section', () => {
expect(shipSkill).toContain('If no verification section or no plan file');
expect(shipSkill).toContain('Automatic diff-scoped QA still runs');
expect(shipSkill).toContain('plan checks beyond that smoke budget remain required');
});
test('skips gracefully when no dev server', () => {
expect(shipSkill).toContain('No dev server detected');
test('blocks unavailable required checks instead of silently skipping them', () => {
const flat = shipSkill.replace(/\s+/g, ' ');
expect(flat).toContain('Noninteractive runs return blocked');
expect(flat).toContain("Send failed, blocked or unrun checks through Step 9's required-probe gate, never silently waive them");
expect(flat).toContain('Missing/unreadable assets block required QA');
expect(flat).toContain('explicitly accept each named probe\'s concrete risk');
expect(flat).toContain('Keep actual outcomes and incomplete flags; VERIFY_RESULT stays fail');
});
});
@@ -1638,9 +1695,9 @@ describe('SPEC_REVIEW_LOOP resolver', () => {
const source = fs.readFileSync(path.join(ROOT, 'plan-ceo-review', 'SKILL.md.tmpl'), 'utf8');
expect(source).toContain('## Plan under review\n{working plan path, or');
const template = source.replace(/\s+/g, ' ');
expect(template).toContain('Prepare the full amended working plan and a separate CEO scope summary');
expect(template).toContain('Keep behavior, requirements and scope consistent');
expect(template).toContain('Prepare the full amended working plan and a separate, consistent CEO scope summary');
expect(template).toContain('the summary cannot serve as the plan');
expect(template).toContain('**Save or present both inputs under the storage policy.**');
});
test('CEO shares both inputs after spec review and owns unresolved concerns in its scope document', () => {
@@ -2446,12 +2503,14 @@ describe('Design approval reconciliation', () => {
const decisions = section.slice(decisionStart, decisionEnd).replace(/\s+/g, ' ');
expect(decisions).toContain('AskUserQuestion({ questions: [currentDecision] })');
expect(decisions).toContain('one question object for one choice; other IDs wait');
expect(decisions).toContain("If you discover another independent choice, return to step 2 before sending the question");
expect(decisions.indexOf("**STOP until the actual answer arrives.**")).toBeLessThan(decisions.indexOf('### 6. Apply and refresh'));
expect(decisions).toContain("Return to step 1 with the updated working plan and answer");
expect(decisions).toContain("If you discover another independent choice, separate it and rebuild this comparison before saving or sending the question");
expect(decisions.indexOf("**STOP until the actual answer arrives.**")).toBeLessThan(decisions.indexOf('### Record the answer'));
expect(decisions).toContain("For the next choice, use the updated working plan and answer");
expect(gate).toContain('report the stale verification and stop');
expect(gate).toContain('starts at Decision procedure for changed choices, then Approval readiness, then repeats affected outputs, Read-back,');
expect(gate).toContain('Review Log and dashboard');
expect(gate).toContain('Resume under **Recovery routing → Late change or missing work**');
const recovery = readSkillUnion('plan-eng-review').split('**Late change or missing work:**')[1]!.split('**Blocked outcome:**')[0]!.replace(/\s+/g, ' ');
expect(recovery).toContain('new or reopened choices use Decision procedure');
expect(recovery).toContain('Repeat Approval readiness, then Required outputs steps 1–4 for changed outputs before choosing navigation again');
expect(gate).toContain('all six columns: Review / Trigger / Why / Runs / Status / Findings');
expect(gate).toContain('follow **Blocked outcome**');
const report = extractMarkdownSection(section, '### Write to the report file');
@@ -4145,7 +4204,7 @@ describe('CONFIDENCE_CALIBRATION resolver', () => {
test(`${skill} generated SKILL.md contains confidence calibration`, () => {
const content = readSkillUnion(skill); // ship: moved to sections/review-army.md
expect(content).toContain('Confidence Calibration');
expect(content).toContain('confidence score');
expect(content).toContain(skill === 'review' ? 'score every finding (1-10)' : 'confidence score');
});
}
@@ -4166,14 +4225,16 @@ describe('CONFIDENCE_CALIBRATION resolver', () => {
test('confidence calibration includes finding format example', () => {
const content = fs.readFileSync(path.join(ROOT, 'review', 'SKILL.md'), 'utf-8');
expect(content).toContain('[P1] (confidence:');
expect(content).toContain('[CRITICAL] (confidence:');
expect(content).toContain('SQL injection');
});
test('confidence calibration includes calibration learning feedback loop', () => {
const content = fs.readFileSync(path.join(ROOT, 'review', 'SKILL.md'), 'utf-8');
expect(content).toContain('calibration event');
expect(content).toContain('Log the corrected pattern');
const flat = content.replace(/\s+/g, ' ');
expect(flat).toContain('Calibration learning');
expect(flat).toContain('If the user confirms a reported finding scored < 7 is real');
expect(flat).toContain('log the corrected pattern as a learning');
});
test('skills without confidence calibration do NOT contain it', () => {
@@ -4460,7 +4521,12 @@ describe('plan-mode-info resolver (handshake-replacement)', () => {
expect(startup).toContain('Before 0E, call 0D for unresolved approaches');
expect(startup).toContain('A) current/requested plan, B) smallest scoped alternative');
expect(startup).toContain('With no required choice, or after those choices settle, go to 0E');
expect(approach).toContain('0D never restarts mode selection');
expect(approach).toContain('0D returns to its caller, not to mode selection');
expect(approach).toContain("For mode changes, follow 0E's **Mode change** instruction");
const modeChange = content.slice(content.indexOf('**Mode change:**'), preludeIdx);
expect(modeChange).toContain('Pause and ask with the four-mode menu; keep the mode until answered');
expect(modeChange).toContain('complete newly applicable Step 0 work in route order, reusing completed work and scope answers');
expect(modeChange).toContain('Then resume the paused step. If unchanged, resume directly');
expect(gate).toContain('Return to the calling step with the saved answer; do not ask it again');
expect(gate).not.toContain("When this step's required decisions are settled, go to 0E if you came from 0C");
expect(gate).toContain('even for a lone option');