mirror of
https://github.com/garrytan/gstack.git
synced 2026-10-04 18:36:54 +02:00
v1.91.7.0 feat: add functional QA and pre-publication docs checks (#2983)
* feat: add surface-aware exploratory QA and ship documentation gates * test: preserve delegated QA setup authority after main integration * fix(qa): clarify exploration order and preserve report artifacts * test(qa): follow the shared setup reference directly * refactor(ship): make verification and recovery routes explicit * test(ship): align evidence and review guards with explicit routes * fix(workflows): clarify ship recovery and functional QA evidence * fix(workflows): clarify approval recovery and full QA coverage * refactor(workflows): order review transactions and clarify ship state * fix(ship): clarify final verification and fail closed at publication * fix(evals): attribute native atomic documentation writes * fix(ship): clarify recovery and documentation lifecycle guidance * fix(test): preserve observed native placeholder styling in CI * fix(codex): report watchdog timeouts without a process-exit race * Checkpoint functional QA implementation and workflow validation repairs * Fix documentation and shared-review fixture contracts * docs: clarify judge reuse and evaluation supervision * test: align review evidence and selected case contracts * test: verify append-only documentation checkpoints and recovery * fix: qualify QA workflows and CI validation repairs * fix: launch shared-libs fixture scripts on Windows * fix: qualify QA deadlines, fixture isolation, and shard cleanup * fix: preserve qualified QA and cancellation repairs * fix: enforce functional fixture authority and share strict event decoding * fix: retain free-test evidence and explain recovery * fix: reject malformed native evidence after decoder consolidation * test: use reliable capture for telemetry privacy filters * test: refresh measured quick coverage and document validation costs * Fix native fixture receipts and preserve VM validation evidence * Align negative judge controls with upstream clarity policy * Fix report-only QA preparation and public evidence handling * Clarify QA-only preparation and current-report preservation * Stream Ship quality judgments with an explicit 64k response contract * Validate compact judge reasoning locally with supported wire schema * Align functional QA fixture instructions with evidence acceptance * Bind native browser diagnostics to execution evidence and align review verdicts * Preserve native diagnostic line boundaries * Serialize functional QA evidence from native captures * Keep large QA evidence fixture payload out of Windows argv
This commit is contained in:
1 parent
65bfb0ce49
commit
dcaea52800
333 files changed
+41755
-7357
No files matched your search
@@ -4,7 +4,7 @@ import { mkdirSync, mkdtempSync, readFileSync, readdirSync, rmSync, writeFileSyn
|
||||
import { tmpdir } from 'node:os';
|
||||
import { dirname, join, resolve } from 'node:path';
|
||||
import { createHash } from 'node:crypto';
|
||||
import { readWorkflowJudgeInput, buildWorkflowJudgePrompt } from './helpers/workflow-judge-input';
|
||||
import { readWorkflowJudgeInput, buildWorkflowJudgePrompt, QA_DISCOVERY_REFERENCES } from './helpers/workflow-judge-input';
|
||||
import { ENG_REVIEW_EXCERPT } from './helpers/workflow-excerpt';
|
||||
|
||||
const ROOT = resolve(import.meta.dir, '..');
|
||||
@@ -18,6 +18,35 @@ test('cache extraction preserves every byte of the original workflow request and
|
||||
expect(createHash('sha256').update(prompt).digest('hex')).toBe('71cc9c777bf28ff0efd610259b411e3539852f83a0888fa0e92469331f8b9a43');
|
||||
});
|
||||
|
||||
test.each(['ship', 'review'])('%s clarity targets frontier readers without excusing missing decisions or authority', skill => {
|
||||
const source = readFileSync(join(ROOT, 'test/skill-llm-eval.test.ts'), 'utf8');
|
||||
const registration = source.match(new RegExp(`testIfSelected\\('${skill}/SKILL\\.md workflow',[\\s\\S]*?await runWorkflowJudge\\(\\{([\\s\\S]*?)\\n \\}\\);`));
|
||||
expect(registration).not.toBeNull();
|
||||
const options = new Function('QA_DISCOVERY_REFERENCES', `return ({${registration![1]}});`)(QA_DISCOVERY_REFERENCES);
|
||||
expect(options.agentCapability).toBe('frontier');
|
||||
expect(options.thresholds).toBeUndefined();
|
||||
const input = readWorkflowJudgeInput({ root: ROOT, ...options });
|
||||
const prompt = buildWorkflowJudgePrompt(options, input);
|
||||
expect(prompt).toContain('GPT-5.6 Sol-level capability or stronger');
|
||||
expect(prompt).toContain('Length, technical vocabulary and multiple explicit recovery paths alone are not clarity defects');
|
||||
expect(prompt).toContain('Do not invent missing policies, permissions or evidence');
|
||||
expect(prompt).toContain('Clarity 4 means the target agent can determine the next permitted action on each applicable path');
|
||||
expect(prompt).toContain('Score clarity 3 or lower when execution still requires guessing');
|
||||
expect(prompt).toContain('conflicting order, undefined decisions, unclear authority or missing input/output handling');
|
||||
expect(prompt).toContain('cite the specific file/step and explain the competing actions or missing decision');
|
||||
expect(prompt.endsWith(input.text)).toBe(true);
|
||||
expect(prompt).toContain('"clarity": N, "completeness": N, "actionability": N, "reasoning": "brief explanation"');
|
||||
});
|
||||
|
||||
test('frontier calibration bounds reporting without reducing the evaluated source bundle', () => {
|
||||
const input = { files: [], text: 'Entire source bundle remains present.' };
|
||||
const prompt = buildWorkflowJudgePrompt({ judgeContext: 'a workflow', judgeGoal: 'how to finish', agentCapability: 'frontier' }, input);
|
||||
expect(prompt).toContain('Evaluate the whole workflow, but keep the JSON reasoning under 150 words with at most two decisive examples');
|
||||
expect(prompt).toContain('For a clarity defect, cite the specific file/step and explain the competing actions or missing decision');
|
||||
expect(prompt).not.toContain('For each clarity defect');
|
||||
expect(prompt.endsWith(input.text)).toBe(true);
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
for (const root of scratchRoots.splice(0)) rmSync(root, { recursive: true, force: true });
|
||||
});
|
||||
@@ -109,6 +138,56 @@ describe('workflow judge file bundle', () => {
|
||||
expect(text).toContain('STOP and read sections/step.md.');
|
||||
});
|
||||
|
||||
test('cross-skill references retain complete bytes once and fail on missing assets', () => {
|
||||
const root = fixture({
|
||||
'example/SKILL.md': '## Begin\nRead the shared resource.\n## End',
|
||||
'example/sections/local.md': 'Complete local section.',
|
||||
'shared/sections/method.md': 'Shared prefix\n## End\nShared suffix',
|
||||
});
|
||||
const options = { root, skillPath: 'example/SKILL.md', startMarker: '## Begin', endMarker: '## End',
|
||||
references: ['example/sections/local.md', 'shared/sections/method.md', 'shared/sections/method.md'] };
|
||||
const input = readWorkflowJudgeInput(options);
|
||||
expect(input.files.filter(file => file.kind === 'reference')).toEqual([
|
||||
{ path: 'shared/sections/method.md', kind: 'reference', content: 'Shared prefix\n## End\nShared suffix', startLine: 1, endLine: 3 },
|
||||
]);
|
||||
expect(input.files.filter(file => file.path === 'example/sections/local.md')).toHaveLength(1);
|
||||
expect(occurrences(input.text, 'Shared prefix')).toBe(1);
|
||||
expect(() => readWorkflowJudgeInput({ ...options, references: ['shared/missing.md'] })).toThrow();
|
||||
expect(() => readWorkflowJudgeInput({ ...options, references: ['../outside.md'] })).toThrow('Reference outside root');
|
||||
});
|
||||
|
||||
for (const skill of ['ship', 'qa-only', 'document-release', 'review']) {
|
||||
test(`actual ${skill} judge includes all referenced QA or documentation sections`, () => {
|
||||
const caller = readFileSync(join(ROOT, 'test/skill-llm-eval.test.ts'), 'utf8');
|
||||
const name = `${skill}/SKILL.md workflow`;
|
||||
const start = caller.indexOf(`testIfSelected('${name}'`);
|
||||
expect(start).toBeGreaterThanOrEqual(0);
|
||||
const registration = caller.slice(start).match(/await runWorkflowJudge\(\{([\s\S]*?)\n \}\);/);
|
||||
expect(registration).not.toBeNull();
|
||||
const options = new Function('QA_DISCOVERY_REFERENCES', `return ({${registration![1]}});`)(QA_DISCOVERY_REFERENCES);
|
||||
const input = readWorkflowJudgeInput({ root: ROOT, ...options });
|
||||
for (const file of options.references ?? []) {
|
||||
expect(input.files.find(item => item.path === file)?.content).toBe(readFileSync(join(ROOT, file), 'utf8'));
|
||||
}
|
||||
if (skill === 'ship') {
|
||||
expect(input.files.map(file => file.path)).toContain('ship/sections/documentation.md');
|
||||
expect(input.files.map(file => file.path)).toContain('qa/sections/exploratory.md');
|
||||
} else if (skill === 'qa-only') {
|
||||
expect(input.files.map(file => file.path)).toContain('qa/sections/system-functional.md');
|
||||
expect(input.files.filter(file => file.path.endsWith('/exploratory.md'))).toHaveLength(1);
|
||||
expect(input.files.find(file => file.kind === 'entrypoint')?.content).toContain('Never fix bugs or write product tests');
|
||||
} else if (skill === 'document-release') {
|
||||
expect(input.files.map(file => file.path)).toContain('document-release/sections/audit-scope.md');
|
||||
expect(input.text).toContain('Ship-owned documentation mode');
|
||||
} else {
|
||||
expect(input.files.map(file => file.path)).toContain('qa/sections/exploratory.md');
|
||||
expect(input.files.map(file => file.path)).toContain('review/checklist.md');
|
||||
expect(input.text).toContain('test_stub');
|
||||
expect(input.text).toContain('## Step 5: Fix-First Review');
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
test('retains section prelude and suffix exactly once when both markers are inside a section', () => {
|
||||
// Generated comments made the old 120-character prefix heuristic append
|
||||
// this whole file after its partial slice, duplicating every review pass.
|
||||
@@ -200,7 +279,16 @@ describe('workflow judge file bundle', () => {
|
||||
const entrypoint = input.files.find(file => file.kind === 'entrypoint');
|
||||
expect(entrypoint?.content).toBe(source.slice(source.indexOf(startMarker), source.indexOf(endMarker, source.indexOf(startMarker))));
|
||||
expect(entrypoint?.content).toContain('git remote get-url origin');
|
||||
expect(entrypoint?.content).toContain('**Follow every STOP and AskUserQuestion gate**');
|
||||
expect(entrypoint?.content).toContain('STOP blocks advancement until the stated repair/resume route clears; without one, end this attempt');
|
||||
const flow = entrypoint!.content.replace(/\s+/g, ' ');
|
||||
expect(flow).toContain('Every new invocation repeats Steps 1–16, including both reviews and the docs audit');
|
||||
expect(flow).toContain('children return evidence, not permission to proceed');
|
||||
expect(flow).toContain('Follow the saved work list');
|
||||
expect(flow).not.toContain('| At step | Outcome |');
|
||||
expect(flow).toContain('`gstack-wtree` prints a Git tree hash');
|
||||
expect(flow).toContain('Offline output without that fallback, failure, malformed output or an empty version is unusable');
|
||||
expect(entrypoint?.content).toContain('Answer each AskUserQuestion before continuing');
|
||||
expect(entrypoint?.content).toContain('Routine authorization never waives those gates or their required user decisions');
|
||||
expect(entrypoint?.content).toContain('## Step 0: Detect platform and base branch');
|
||||
expect(entrypoint?.content).toContain('gh pr view --json baseRefName');
|
||||
expect(entrypoint?.content).toContain('Print the detected base branch name.');
|
||||
|
||||
Reference in new issue
Block a user