Files
gstack/test/helpers/workflow-judge-input.ts
T
Garry Tan dcaea52800 v1.91.7.0 feat: add functional QA and pre-publication docs checks (#2983)
* feat: add surface-aware exploratory QA and ship documentation gates

* test: preserve delegated QA setup authority after main integration

* fix(qa): clarify exploration order and preserve report artifacts

* test(qa): follow the shared setup reference directly

* refactor(ship): make verification and recovery routes explicit

* test(ship): align evidence and review guards with explicit routes

* fix(workflows): clarify ship recovery and functional QA evidence

* fix(workflows): clarify approval recovery and full QA coverage

* refactor(workflows): order review transactions and clarify ship state

* fix(ship): clarify final verification and fail closed at publication

* fix(evals): attribute native atomic documentation writes

* fix(ship): clarify recovery and documentation lifecycle guidance

* fix(test): preserve observed native placeholder styling in CI

* fix(codex): report watchdog timeouts without a process-exit race

* Checkpoint functional QA implementation and workflow validation repairs

* Fix documentation and shared-review fixture contracts

* docs: clarify judge reuse and evaluation supervision

* test: align review evidence and selected case contracts

* test: verify append-only documentation checkpoints and recovery

* fix: qualify QA workflows and CI validation repairs

* fix: launch shared-libs fixture scripts on Windows

* fix: qualify QA deadlines, fixture isolation, and shard cleanup

* fix: preserve qualified QA and cancellation repairs

* fix: enforce functional fixture authority and share strict event decoding

* fix: retain free-test evidence and explain recovery

* fix: reject malformed native evidence after decoder consolidation

* test: use reliable capture for telemetry privacy filters

* test: refresh measured quick coverage and document validation costs

* Fix native fixture receipts and preserve VM validation evidence

* Align negative judge controls with upstream clarity policy

* Fix report-only QA preparation and public evidence handling

* Clarify QA-only preparation and current-report preservation

* Stream Ship quality judgments with an explicit 64k response contract

* Validate compact judge reasoning locally with supported wire schema

* Align functional QA fixture instructions with evidence acceptance

* Bind native browser diagnostics to execution evidence and align review verdicts

* Preserve native diagnostic line boundaries

* Serialize functional QA evidence from native captures

* Keep large QA evidence fixture payload out of Windows argv
2026-09-29 06:07:35 -07:00

156 lines
7.1 KiB
TypeScript

/** Preserve source-file boundaries when a workflow judge reads carved skills. */
import * as fs from 'node:fs';
import * as path from 'node:path';
export interface WorkflowJudgeFile {
path: string;
kind: 'entrypoint' | 'section' | 'reference';
content: string;
startLine: number;
endLine: number;
}
export interface WorkflowJudgeInput {
files: WorkflowJudgeFile[];
text: string;
}
export const QA_DISCOVERY_REFERENCES = [
'qa/sections/scope.md',
'qa/sections/exploratory.md',
'qa/sections/system-functional.md',
'qa/sections/browser-setup.md',
'qa/sections/qa-patterns.md',
'qa/templates/functional-report-template.md',
];
export const WORKFLOW_JUDGE_REASONING_WORD_LIMIT = 150;
export const WORKFLOW_JUDGE_RESPONSE_SCHEMA = {
type: 'object',
properties: {
clarity: { type: 'integer', enum: [1, 2, 3, 4, 5] },
completeness: { type: 'integer', enum: [1, 2, 3, 4, 5] },
actionability: { type: 'integer', enum: [1, 2, 3, 4, 5] },
reasoning: { type: 'string',
description: `Under ${WORKFLOW_JUDGE_REASONING_WORD_LIMIT} words with at most two decisive examples, evaluating the complete supplied workflow.` },
},
required: ['clarity', 'completeness', 'actionability', 'reasoning'],
additionalProperties: false,
};
export function buildWorkflowJudgePrompt(opts: {
judgeContext: string;
judgeGoal: string;
agentCapability?: 'frontier';
}, input: WorkflowJudgeInput): string {
return `You are evaluating the quality of ${opts.judgeContext} for an AI coding agent.
The agent reads these source files to learn ${opts.judgeGoal}. Shared preamble definitions and
external tools/files are documented separately; do not penalize their absence from this bundle.
On-demand sections retain their original file boundaries and Read instructions; the section
index refers to those files, not duplicate work. The bundle order is not execution order.
Judge the actual instructions, including contradictory ordering or missing decisions.${opts.agentCapability === 'frontier' ? `
Target reader: a frontier coding agent with GPT-5.6 Sol-level capability or stronger.
Assume it can follow explicit cross-references, track saved state and a bounded work list,
and distinguish conditional branches. Length, technical vocabulary and multiple explicit recovery paths alone are not clarity defects.
Do not invent missing policies, permissions or evidence to make a workflow executable.
Clarity 4 means the target agent can determine the next permitted action on each applicable path;
5 additionally means those paths are easy to locate and understand.
Score clarity 3 or lower when execution still requires guessing because of
conflicting order, undefined decisions, unclear authority or missing input/output handling.
Evaluate the whole workflow, but keep the JSON reasoning under 150 words with at most two decisive examples.
For a clarity defect, cite the specific file/step and explain the competing actions or missing decision.
Keep completeness and actionability independent: reader capability does not supply missing requirements.` : ''}
Rate on three dimensions (1-5 scale):
- **clarity** (1-5): Can an agent follow the instructions without ambiguity?
- **completeness** (1-5): Are all steps, decision points, and outputs well-defined?
- **actionability** (1-5): Can an agent execute this workflow and produce the expected deliverables?
Respond with ONLY valid JSON:
{"clarity": N, "completeness": N, "actionability": N, "reasoning": "brief explanation"}
Here is the source-file bundle to evaluate:
${input.text}`;
}
export function readWorkflowJudgeInput(opts: {
root: string;
skillPath: string;
startMarker: string;
endMarker: string | null;
references?: readonly string[];
}): WorkflowJudgeInput {
const sources = [{
path: opts.skillPath,
kind: 'entrypoint' as const,
content: fs.readFileSync(path.join(opts.root, opts.skillPath), 'utf8'),
}];
const sectionDir = path.join(path.dirname(opts.skillPath), 'sections');
const sectionRoot = path.join(opts.root, sectionDir);
const sections = fs.existsSync(sectionRoot)
? fs.readdirSync(sectionRoot).sort().filter(name => name.endsWith('.md'))
.map(name => ({
path: path.join(sectionDir, name),
kind: 'section' as const,
content: fs.readFileSync(path.join(sectionRoot, name), 'utf8'),
}))
: [];
const references = [...new Set(opts.references ?? [])].map(file => {
const resolved = path.resolve(opts.root, file);
if (!resolved.startsWith(path.resolve(opts.root) + path.sep)) throw new Error(`Reference outside root: ${file}`);
return { path: file, kind: 'reference' as const, content: fs.readFileSync(resolved, 'utf8') };
}).filter(file => ![...sources, ...sections].some(source => source.path === file.path));
const allSources = [...sources, ...sections, ...references];
// Preserve the existing marker window, including markers that moved into
// section files. Offsets identify the source of each slice; prose prefixes
// cannot reliably identify a file after its generated header was sliced off.
const union = allSources.map(file => file.content).join('\n');
const start = union.indexOf(opts.startMarker);
if (start < 0) throw new Error(`Start marker not found in ${opts.skillPath}: "${opts.startMarker}"`);
const end = opts.endMarker === null ? union.length : union.indexOf(opts.endMarker, start);
if (end < 0) throw new Error(`End marker not found in ${opts.skillPath}: "${opts.endMarker}"`);
const files: WorkflowJudgeFile[] = [];
let offset = 0;
for (const file of allSources) {
// Every section was already supplied in full by the old judge input. Keep
// that coverage, but include each file once even when the marker window
// also covers part of it. Only the entrypoint retains the requested slice.
const from = file.kind !== 'entrypoint' ? 0 : Math.max(0, start - offset);
const to = file.kind !== 'entrypoint' ? file.content.length : Math.min(file.content.length, end - offset);
if (from < to) {
files.push({
...file,
path: file.path.split(path.sep).join('/'),
content: file.content.slice(from, to),
startLine: file.content.slice(0, from).split('\n').length,
endLine: file.content.slice(0, to - 1).split('\n').length,
});
}
offset += file.content.length + 1;
}
const context = [
'The material below is a bundle of source-file excerpts, with each original file and line range labeled.',
'SKILL.md is the entry point; the labeled ranges identify which excerpts are supplied.',
...(sections.length + references.length > 0 ? [
'The section files remain separate on disk and are read at the points and conditions specified by the skill\'s Read directives.',
'They are supplied here as on-demand references; their order in this bundle is not execution order.',
] : []),
].join('\n');
return {
files,
text: context + '\n\n' + files.map(file => [
`--- BEGIN FILE ${JSON.stringify(file.path)} (lines ${file.startLine}-${file.endLine}; ${file.kind}) ---`,
file.content,
`--- END FILE ${JSON.stringify(file.path)} ---`,
].join('\n')).join('\n\n'),
};
}