v1.91.7.0 feat: add functional QA and pre-publication docs checks (#2983)

* feat: add surface-aware exploratory QA and ship documentation gates

* test: preserve delegated QA setup authority after main integration

* fix(qa): clarify exploration order and preserve report artifacts

* test(qa): follow the shared setup reference directly

* refactor(ship): make verification and recovery routes explicit

* test(ship): align evidence and review guards with explicit routes

* fix(workflows): clarify ship recovery and functional QA evidence

* fix(workflows): clarify approval recovery and full QA coverage

* refactor(workflows): order review transactions and clarify ship state

* fix(ship): clarify final verification and fail closed at publication

* fix(evals): attribute native atomic documentation writes

* fix(ship): clarify recovery and documentation lifecycle guidance

* fix(test): preserve observed native placeholder styling in CI

* fix(codex): report watchdog timeouts without a process-exit race

* Checkpoint functional QA implementation and workflow validation repairs

* Fix documentation and shared-review fixture contracts

* docs: clarify judge reuse and evaluation supervision

* test: align review evidence and selected case contracts

* test: verify append-only documentation checkpoints and recovery

* fix: qualify QA workflows and CI validation repairs

* fix: launch shared-libs fixture scripts on Windows

* fix: qualify QA deadlines, fixture isolation, and shard cleanup

* fix: preserve qualified QA and cancellation repairs

* fix: enforce functional fixture authority and share strict event decoding

* fix: retain free-test evidence and explain recovery

* fix: reject malformed native evidence after decoder consolidation

* test: use reliable capture for telemetry privacy filters

* test: refresh measured quick coverage and document validation costs

* Fix native fixture receipts and preserve VM validation evidence

* Align negative judge controls with upstream clarity policy

* Fix report-only QA preparation and public evidence handling

* Clarify QA-only preparation and current-report preservation

* Stream Ship quality judgments with an explicit 64k response contract

* Validate compact judge reasoning locally with supported wire schema

* Align functional QA fixture instructions with evidence acceptance

* Bind native browser diagnostics to execution evidence and align review verdicts

* Preserve native diagnostic line boundaries

* Serialize functional QA evidence from native captures

* Keep large QA evidence fixture payload out of Windows argv
This commit is contained in:
Garry Tan authored and GitHub committed 2026-09-29 06:07:35 -07:00
1 parent 65bfb0ce49
commit dcaea52800
333 files changed
+41755 -7357

No files matched your search

+90 -2
View File
@@ -4,7 +4,7 @@ import { mkdirSync, mkdtempSync, readFileSync, readdirSync, rmSync, writeFileSyn
import { tmpdir } from 'node:os';
import { dirname, join, resolve } from 'node:path';
import { createHash } from 'node:crypto';
import { readWorkflowJudgeInput, buildWorkflowJudgePrompt } from './helpers/workflow-judge-input';
import { readWorkflowJudgeInput, buildWorkflowJudgePrompt, QA_DISCOVERY_REFERENCES } from './helpers/workflow-judge-input';
import { ENG_REVIEW_EXCERPT } from './helpers/workflow-excerpt';
const ROOT = resolve(import.meta.dir, '..');
@@ -18,6 +18,35 @@ test('cache extraction preserves every byte of the original workflow request and
expect(createHash('sha256').update(prompt).digest('hex')).toBe('71cc9c777bf28ff0efd610259b411e3539852f83a0888fa0e92469331f8b9a43');
});
test.each(['ship', 'review'])('%s clarity targets frontier readers without excusing missing decisions or authority', skill => {
const source = readFileSync(join(ROOT, 'test/skill-llm-eval.test.ts'), 'utf8');
const registration = source.match(new RegExp(`testIfSelected\\('${skill}/SKILL\\.md workflow',[\\s\\S]*?await runWorkflowJudge\\(\\{([\\s\\S]*?)\\n \\}\\);`));
expect(registration).not.toBeNull();
const options = new Function('QA_DISCOVERY_REFERENCES', `return ({${registration![1]}});`)(QA_DISCOVERY_REFERENCES);
expect(options.agentCapability).toBe('frontier');
expect(options.thresholds).toBeUndefined();
const input = readWorkflowJudgeInput({ root: ROOT, ...options });
const prompt = buildWorkflowJudgePrompt(options, input);
expect(prompt).toContain('GPT-5.6 Sol-level capability or stronger');
expect(prompt).toContain('Length, technical vocabulary and multiple explicit recovery paths alone are not clarity defects');
expect(prompt).toContain('Do not invent missing policies, permissions or evidence');
expect(prompt).toContain('Clarity 4 means the target agent can determine the next permitted action on each applicable path');
expect(prompt).toContain('Score clarity 3 or lower when execution still requires guessing');
expect(prompt).toContain('conflicting order, undefined decisions, unclear authority or missing input/output handling');
expect(prompt).toContain('cite the specific file/step and explain the competing actions or missing decision');
expect(prompt.endsWith(input.text)).toBe(true);
expect(prompt).toContain('"clarity": N, "completeness": N, "actionability": N, "reasoning": "brief explanation"');
});
test('frontier calibration bounds reporting without reducing the evaluated source bundle', () => {
const input = { files: [], text: 'Entire source bundle remains present.' };
const prompt = buildWorkflowJudgePrompt({ judgeContext: 'a workflow', judgeGoal: 'how to finish', agentCapability: 'frontier' }, input);
expect(prompt).toContain('Evaluate the whole workflow, but keep the JSON reasoning under 150 words with at most two decisive examples');
expect(prompt).toContain('For a clarity defect, cite the specific file/step and explain the competing actions or missing decision');
expect(prompt).not.toContain('For each clarity defect');
expect(prompt.endsWith(input.text)).toBe(true);
});
afterEach(() => {
for (const root of scratchRoots.splice(0)) rmSync(root, { recursive: true, force: true });
});
@@ -109,6 +138,56 @@ describe('workflow judge file bundle', () => {
expect(text).toContain('STOP and read sections/step.md.');
});
test('cross-skill references retain complete bytes once and fail on missing assets', () => {
const root = fixture({
'example/SKILL.md': '## Begin\nRead the shared resource.\n## End',
'example/sections/local.md': 'Complete local section.',
'shared/sections/method.md': 'Shared prefix\n## End\nShared suffix',
});
const options = { root, skillPath: 'example/SKILL.md', startMarker: '## Begin', endMarker: '## End',
references: ['example/sections/local.md', 'shared/sections/method.md', 'shared/sections/method.md'] };
const input = readWorkflowJudgeInput(options);
expect(input.files.filter(file => file.kind === 'reference')).toEqual([
{ path: 'shared/sections/method.md', kind: 'reference', content: 'Shared prefix\n## End\nShared suffix', startLine: 1, endLine: 3 },
]);
expect(input.files.filter(file => file.path === 'example/sections/local.md')).toHaveLength(1);
expect(occurrences(input.text, 'Shared prefix')).toBe(1);
expect(() => readWorkflowJudgeInput({ ...options, references: ['shared/missing.md'] })).toThrow();
expect(() => readWorkflowJudgeInput({ ...options, references: ['../outside.md'] })).toThrow('Reference outside root');
});
for (const skill of ['ship', 'qa-only', 'document-release', 'review']) {
test(`actual ${skill} judge includes all referenced QA or documentation sections`, () => {
const caller = readFileSync(join(ROOT, 'test/skill-llm-eval.test.ts'), 'utf8');
const name = `${skill}/SKILL.md workflow`;
const start = caller.indexOf(`testIfSelected('${name}'`);
expect(start).toBeGreaterThanOrEqual(0);
const registration = caller.slice(start).match(/await runWorkflowJudge\(\{([\s\S]*?)\n \}\);/);
expect(registration).not.toBeNull();
const options = new Function('QA_DISCOVERY_REFERENCES', `return ({${registration![1]}});`)(QA_DISCOVERY_REFERENCES);
const input = readWorkflowJudgeInput({ root: ROOT, ...options });
for (const file of options.references ?? []) {
expect(input.files.find(item => item.path === file)?.content).toBe(readFileSync(join(ROOT, file), 'utf8'));
}
if (skill === 'ship') {
expect(input.files.map(file => file.path)).toContain('ship/sections/documentation.md');
expect(input.files.map(file => file.path)).toContain('qa/sections/exploratory.md');
} else if (skill === 'qa-only') {
expect(input.files.map(file => file.path)).toContain('qa/sections/system-functional.md');
expect(input.files.filter(file => file.path.endsWith('/exploratory.md'))).toHaveLength(1);
expect(input.files.find(file => file.kind === 'entrypoint')?.content).toContain('Never fix bugs or write product tests');
} else if (skill === 'document-release') {
expect(input.files.map(file => file.path)).toContain('document-release/sections/audit-scope.md');
expect(input.text).toContain('Ship-owned documentation mode');
} else {
expect(input.files.map(file => file.path)).toContain('qa/sections/exploratory.md');
expect(input.files.map(file => file.path)).toContain('review/checklist.md');
expect(input.text).toContain('test_stub');
expect(input.text).toContain('## Step 5: Fix-First Review');
}
});
}
test('retains section prelude and suffix exactly once when both markers are inside a section', () => {
// Generated comments made the old 120-character prefix heuristic append
// this whole file after its partial slice, duplicating every review pass.
@@ -200,7 +279,16 @@ describe('workflow judge file bundle', () => {
const entrypoint = input.files.find(file => file.kind === 'entrypoint');
expect(entrypoint?.content).toBe(source.slice(source.indexOf(startMarker), source.indexOf(endMarker, source.indexOf(startMarker))));
expect(entrypoint?.content).toContain('git remote get-url origin');
expect(entrypoint?.content).toContain('**Follow every STOP and AskUserQuestion gate**');
expect(entrypoint?.content).toContain('STOP blocks advancement until the stated repair/resume route clears; without one, end this attempt');
const flow = entrypoint!.content.replace(/\s+/g, ' ');
expect(flow).toContain('Every new invocation repeats Steps 1–16, including both reviews and the docs audit');
expect(flow).toContain('children return evidence, not permission to proceed');
expect(flow).toContain('Follow the saved work list');
expect(flow).not.toContain('| At step | Outcome |');
expect(flow).toContain('`gstack-wtree` prints a Git tree hash');
expect(flow).toContain('Offline output without that fallback, failure, malformed output or an empty version is unusable');
expect(entrypoint?.content).toContain('Answer each AskUserQuestion before continuing');
expect(entrypoint?.content).toContain('Routine authorization never waives those gates or their required user decisions');
expect(entrypoint?.content).toContain('## Step 0: Detect platform and base branch');
expect(entrypoint?.content).toContain('gh pr view --json baseRefName');
expect(entrypoint?.content).toContain('Print the detected base branch name.');