mirror of
https://github.com/garrytan/gstack.git
synced 2026-10-08 20:31:45 +02:00
v1.91.7.0 feat: add functional QA and pre-publication docs checks (#2983)
* feat: add surface-aware exploratory QA and ship documentation gates * test: preserve delegated QA setup authority after main integration * fix(qa): clarify exploration order and preserve report artifacts * test(qa): follow the shared setup reference directly * refactor(ship): make verification and recovery routes explicit * test(ship): align evidence and review guards with explicit routes * fix(workflows): clarify ship recovery and functional QA evidence * fix(workflows): clarify approval recovery and full QA coverage * refactor(workflows): order review transactions and clarify ship state * fix(ship): clarify final verification and fail closed at publication * fix(evals): attribute native atomic documentation writes * fix(ship): clarify recovery and documentation lifecycle guidance * fix(test): preserve observed native placeholder styling in CI * fix(codex): report watchdog timeouts without a process-exit race * Checkpoint functional QA implementation and workflow validation repairs * Fix documentation and shared-review fixture contracts * docs: clarify judge reuse and evaluation supervision * test: align review evidence and selected case contracts * test: verify append-only documentation checkpoints and recovery * fix: qualify QA workflows and CI validation repairs * fix: launch shared-libs fixture scripts on Windows * fix: qualify QA deadlines, fixture isolation, and shard cleanup * fix: preserve qualified QA and cancellation repairs * fix: enforce functional fixture authority and share strict event decoding * fix: retain free-test evidence and explain recovery * fix: reject malformed native evidence after decoder consolidation * test: use reliable capture for telemetry privacy filters * test: refresh measured quick coverage and document validation costs * Fix native fixture receipts and preserve VM validation evidence * Align negative judge controls with upstream clarity policy * Fix report-only QA preparation and public evidence handling * Clarify QA-only preparation and current-report preservation * Stream Ship quality judgments with an explicit 64k response contract * Validate compact judge reasoning locally with supported wire schema * Align functional QA fixture instructions with evidence acceptance * Bind native browser diagnostics to execution evidence and align review verdicts * Preserve native diagnostic line boundaries * Serialize functional QA evidence from native captures * Keep large QA evidence fixture payload out of Windows argv
This commit is contained in:
1 parent
65bfb0ce49
commit
dcaea52800
333 files changed
+41755
-7357
No files matched your search
@@ -4,7 +4,7 @@ import * as path from 'node:path';
|
||||
|
||||
export interface WorkflowJudgeFile {
|
||||
path: string;
|
||||
kind: 'entrypoint' | 'section';
|
||||
kind: 'entrypoint' | 'section' | 'reference';
|
||||
content: string;
|
||||
startLine: number;
|
||||
endLine: number;
|
||||
@@ -15,15 +15,55 @@ export interface WorkflowJudgeInput {
|
||||
text: string;
|
||||
}
|
||||
|
||||
/** Exact existing rubric/request text; extraction must not resample a new prompt. */
|
||||
export function buildWorkflowJudgePrompt(opts: { judgeContext: string; judgeGoal: string }, input: WorkflowJudgeInput): string {
|
||||
export const QA_DISCOVERY_REFERENCES = [
|
||||
'qa/sections/scope.md',
|
||||
'qa/sections/exploratory.md',
|
||||
'qa/sections/system-functional.md',
|
||||
'qa/sections/browser-setup.md',
|
||||
'qa/sections/qa-patterns.md',
|
||||
'qa/templates/functional-report-template.md',
|
||||
];
|
||||
|
||||
export const WORKFLOW_JUDGE_REASONING_WORD_LIMIT = 150;
|
||||
|
||||
export const WORKFLOW_JUDGE_RESPONSE_SCHEMA = {
|
||||
type: 'object',
|
||||
properties: {
|
||||
clarity: { type: 'integer', enum: [1, 2, 3, 4, 5] },
|
||||
completeness: { type: 'integer', enum: [1, 2, 3, 4, 5] },
|
||||
actionability: { type: 'integer', enum: [1, 2, 3, 4, 5] },
|
||||
reasoning: { type: 'string',
|
||||
description: `Under ${WORKFLOW_JUDGE_REASONING_WORD_LIMIT} words with at most two decisive examples, evaluating the complete supplied workflow.` },
|
||||
},
|
||||
required: ['clarity', 'completeness', 'actionability', 'reasoning'],
|
||||
additionalProperties: false,
|
||||
};
|
||||
|
||||
export function buildWorkflowJudgePrompt(opts: {
|
||||
judgeContext: string;
|
||||
judgeGoal: string;
|
||||
agentCapability?: 'frontier';
|
||||
}, input: WorkflowJudgeInput): string {
|
||||
return `You are evaluating the quality of ${opts.judgeContext} for an AI coding agent.
|
||||
|
||||
The agent reads these source files to learn ${opts.judgeGoal}. Shared preamble definitions and
|
||||
external tools/files are documented separately; do not penalize their absence from this bundle.
|
||||
On-demand sections retain their original file boundaries and Read instructions; the section
|
||||
index refers to those files, not duplicate work. The bundle order is not execution order.
|
||||
Judge the actual instructions, including contradictory ordering or missing decisions.
|
||||
Judge the actual instructions, including contradictory ordering or missing decisions.${opts.agentCapability === 'frontier' ? `
|
||||
|
||||
Target reader: a frontier coding agent with GPT-5.6 Sol-level capability or stronger.
|
||||
Assume it can follow explicit cross-references, track saved state and a bounded work list,
|
||||
and distinguish conditional branches. Length, technical vocabulary and multiple explicit recovery paths alone are not clarity defects.
|
||||
Do not invent missing policies, permissions or evidence to make a workflow executable.
|
||||
|
||||
Clarity 4 means the target agent can determine the next permitted action on each applicable path;
|
||||
5 additionally means those paths are easy to locate and understand.
|
||||
Score clarity 3 or lower when execution still requires guessing because of
|
||||
conflicting order, undefined decisions, unclear authority or missing input/output handling.
|
||||
Evaluate the whole workflow, but keep the JSON reasoning under 150 words with at most two decisive examples.
|
||||
For a clarity defect, cite the specific file/step and explain the competing actions or missing decision.
|
||||
Keep completeness and actionability independent: reader capability does not supply missing requirements.` : ''}
|
||||
|
||||
Rate on three dimensions (1-5 scale):
|
||||
- **clarity** (1-5): Can an agent follow the instructions without ambiguity?
|
||||
@@ -43,6 +83,7 @@ export function readWorkflowJudgeInput(opts: {
|
||||
skillPath: string;
|
||||
startMarker: string;
|
||||
endMarker: string | null;
|
||||
references?: readonly string[];
|
||||
}): WorkflowJudgeInput {
|
||||
const sources = [{
|
||||
path: opts.skillPath,
|
||||
@@ -59,7 +100,12 @@ export function readWorkflowJudgeInput(opts: {
|
||||
content: fs.readFileSync(path.join(sectionRoot, name), 'utf8'),
|
||||
}))
|
||||
: [];
|
||||
const allSources = [...sources, ...sections];
|
||||
const references = [...new Set(opts.references ?? [])].map(file => {
|
||||
const resolved = path.resolve(opts.root, file);
|
||||
if (!resolved.startsWith(path.resolve(opts.root) + path.sep)) throw new Error(`Reference outside root: ${file}`);
|
||||
return { path: file, kind: 'reference' as const, content: fs.readFileSync(resolved, 'utf8') };
|
||||
}).filter(file => ![...sources, ...sections].some(source => source.path === file.path));
|
||||
const allSources = [...sources, ...sections, ...references];
|
||||
|
||||
// Preserve the existing marker window, including markers that moved into
|
||||
// section files. Offsets identify the source of each slice; prose prefixes
|
||||
@@ -76,8 +122,8 @@ export function readWorkflowJudgeInput(opts: {
|
||||
// Every section was already supplied in full by the old judge input. Keep
|
||||
// that coverage, but include each file once even when the marker window
|
||||
// also covers part of it. Only the entrypoint retains the requested slice.
|
||||
const from = file.kind === 'section' ? 0 : Math.max(0, start - offset);
|
||||
const to = file.kind === 'section' ? file.content.length : Math.min(file.content.length, end - offset);
|
||||
const from = file.kind !== 'entrypoint' ? 0 : Math.max(0, start - offset);
|
||||
const to = file.kind !== 'entrypoint' ? file.content.length : Math.min(file.content.length, end - offset);
|
||||
if (from < to) {
|
||||
files.push({
|
||||
...file,
|
||||
@@ -93,7 +139,7 @@ export function readWorkflowJudgeInput(opts: {
|
||||
const context = [
|
||||
'The material below is a bundle of source-file excerpts, with each original file and line range labeled.',
|
||||
'SKILL.md is the entry point; the labeled ranges identify which excerpts are supplied.',
|
||||
...(sections.length > 0 ? [
|
||||
...(sections.length + references.length > 0 ? [
|
||||
'The section files remain separate on disk and are read at the points and conditions specified by the skill\'s Read directives.',
|
||||
'They are supplied here as on-demand references; their order in this bundle is not execution order.',
|
||||
] : []),
|
||||
|
||||
Reference in new issue
Block a user