v1.91.7.0 feat: add functional QA and pre-publication docs checks (#2983)

* feat: add surface-aware exploratory QA and ship documentation gates

* test: preserve delegated QA setup authority after main integration

* fix(qa): clarify exploration order and preserve report artifacts

* test(qa): follow the shared setup reference directly

* refactor(ship): make verification and recovery routes explicit

* test(ship): align evidence and review guards with explicit routes

* fix(workflows): clarify ship recovery and functional QA evidence

* fix(workflows): clarify approval recovery and full QA coverage

* refactor(workflows): order review transactions and clarify ship state

* fix(ship): clarify final verification and fail closed at publication

* fix(evals): attribute native atomic documentation writes

* fix(ship): clarify recovery and documentation lifecycle guidance

* fix(test): preserve observed native placeholder styling in CI

* fix(codex): report watchdog timeouts without a process-exit race

* Checkpoint functional QA implementation and workflow validation repairs

* Fix documentation and shared-review fixture contracts

* docs: clarify judge reuse and evaluation supervision

* test: align review evidence and selected case contracts

* test: verify append-only documentation checkpoints and recovery

* fix: qualify QA workflows and CI validation repairs

* fix: launch shared-libs fixture scripts on Windows

* fix: qualify QA deadlines, fixture isolation, and shard cleanup

* fix: preserve qualified QA and cancellation repairs

* fix: enforce functional fixture authority and share strict event decoding

* fix: retain free-test evidence and explain recovery

* fix: reject malformed native evidence after decoder consolidation

* test: use reliable capture for telemetry privacy filters

* test: refresh measured quick coverage and document validation costs

* Fix native fixture receipts and preserve VM validation evidence

* Align negative judge controls with upstream clarity policy

* Fix report-only QA preparation and public evidence handling

* Clarify QA-only preparation and current-report preservation

* Stream Ship quality judgments with an explicit 64k response contract

* Validate compact judge reasoning locally with supported wire schema

* Align functional QA fixture instructions with evidence acceptance

* Bind native browser diagnostics to execution evidence and align review verdicts

* Preserve native diagnostic line boundaries

* Serialize functional QA evidence from native captures

* Keep large QA evidence fixture payload out of Windows argv
This commit is contained in:
Garry Tan authored and GitHub committed 2026-09-29 06:07:35 -07:00
1 parent 65bfb0ce49
commit dcaea52800
333 files changed
+41755 -7357

No files matched your search

+54 -8
View File
@@ -4,7 +4,7 @@ import * as path from 'node:path';
export interface WorkflowJudgeFile {
path: string;
kind: 'entrypoint' | 'section';
kind: 'entrypoint' | 'section' | 'reference';
content: string;
startLine: number;
endLine: number;
@@ -15,15 +15,55 @@ export interface WorkflowJudgeInput {
text: string;
}
/** Exact existing rubric/request text; extraction must not resample a new prompt. */
export function buildWorkflowJudgePrompt(opts: { judgeContext: string; judgeGoal: string }, input: WorkflowJudgeInput): string {
export const QA_DISCOVERY_REFERENCES = [
'qa/sections/scope.md',
'qa/sections/exploratory.md',
'qa/sections/system-functional.md',
'qa/sections/browser-setup.md',
'qa/sections/qa-patterns.md',
'qa/templates/functional-report-template.md',
];
export const WORKFLOW_JUDGE_REASONING_WORD_LIMIT = 150;
export const WORKFLOW_JUDGE_RESPONSE_SCHEMA = {
type: 'object',
properties: {
clarity: { type: 'integer', enum: [1, 2, 3, 4, 5] },
completeness: { type: 'integer', enum: [1, 2, 3, 4, 5] },
actionability: { type: 'integer', enum: [1, 2, 3, 4, 5] },
reasoning: { type: 'string',
description: `Under ${WORKFLOW_JUDGE_REASONING_WORD_LIMIT} words with at most two decisive examples, evaluating the complete supplied workflow.` },
},
required: ['clarity', 'completeness', 'actionability', 'reasoning'],
additionalProperties: false,
};
export function buildWorkflowJudgePrompt(opts: {
judgeContext: string;
judgeGoal: string;
agentCapability?: 'frontier';
}, input: WorkflowJudgeInput): string {
return `You are evaluating the quality of ${opts.judgeContext} for an AI coding agent.
The agent reads these source files to learn ${opts.judgeGoal}. Shared preamble definitions and
external tools/files are documented separately; do not penalize their absence from this bundle.
On-demand sections retain their original file boundaries and Read instructions; the section
index refers to those files, not duplicate work. The bundle order is not execution order.
Judge the actual instructions, including contradictory ordering or missing decisions.
Judge the actual instructions, including contradictory ordering or missing decisions.${opts.agentCapability === 'frontier' ? `
Target reader: a frontier coding agent with GPT-5.6 Sol-level capability or stronger.
Assume it can follow explicit cross-references, track saved state and a bounded work list,
and distinguish conditional branches. Length, technical vocabulary and multiple explicit recovery paths alone are not clarity defects.
Do not invent missing policies, permissions or evidence to make a workflow executable.
Clarity 4 means the target agent can determine the next permitted action on each applicable path;
5 additionally means those paths are easy to locate and understand.
Score clarity 3 or lower when execution still requires guessing because of
conflicting order, undefined decisions, unclear authority or missing input/output handling.
Evaluate the whole workflow, but keep the JSON reasoning under 150 words with at most two decisive examples.
For a clarity defect, cite the specific file/step and explain the competing actions or missing decision.
Keep completeness and actionability independent: reader capability does not supply missing requirements.` : ''}
Rate on three dimensions (1-5 scale):
- **clarity** (1-5): Can an agent follow the instructions without ambiguity?
@@ -43,6 +83,7 @@ export function readWorkflowJudgeInput(opts: {
skillPath: string;
startMarker: string;
endMarker: string | null;
references?: readonly string[];
}): WorkflowJudgeInput {
const sources = [{
path: opts.skillPath,
@@ -59,7 +100,12 @@ export function readWorkflowJudgeInput(opts: {
content: fs.readFileSync(path.join(sectionRoot, name), 'utf8'),
}))
: [];
const allSources = [...sources, ...sections];
const references = [...new Set(opts.references ?? [])].map(file => {
const resolved = path.resolve(opts.root, file);
if (!resolved.startsWith(path.resolve(opts.root) + path.sep)) throw new Error(`Reference outside root: ${file}`);
return { path: file, kind: 'reference' as const, content: fs.readFileSync(resolved, 'utf8') };
}).filter(file => ![...sources, ...sections].some(source => source.path === file.path));
const allSources = [...sources, ...sections, ...references];
// Preserve the existing marker window, including markers that moved into
// section files. Offsets identify the source of each slice; prose prefixes
@@ -76,8 +122,8 @@ export function readWorkflowJudgeInput(opts: {
// Every section was already supplied in full by the old judge input. Keep
// that coverage, but include each file once even when the marker window
// also covers part of it. Only the entrypoint retains the requested slice.
const from = file.kind === 'section' ? 0 : Math.max(0, start - offset);
const to = file.kind === 'section' ? file.content.length : Math.min(file.content.length, end - offset);
const from = file.kind !== 'entrypoint' ? 0 : Math.max(0, start - offset);
const to = file.kind !== 'entrypoint' ? file.content.length : Math.min(file.content.length, end - offset);
if (from < to) {
files.push({
...file,
@@ -93,7 +139,7 @@ export function readWorkflowJudgeInput(opts: {
const context = [
'The material below is a bundle of source-file excerpts, with each original file and line range labeled.',
'SKILL.md is the entry point; the labeled ranges identify which excerpts are supplied.',
...(sections.length > 0 ? [
...(sections.length + references.length > 0 ? [
'The section files remain separate on disk and are read at the points and conditions specified by the skill\'s Read directives.',
'They are supplied here as on-demand references; their order in this bundle is not execution order.',
] : []),