mirror of
https://github.com/garrytan/gstack.git
synced 2026-10-03 01:46:55 +02:00
* feat: add surface-aware exploratory QA and ship documentation gates * test: preserve delegated QA setup authority after main integration * fix(qa): clarify exploration order and preserve report artifacts * test(qa): follow the shared setup reference directly * refactor(ship): make verification and recovery routes explicit * test(ship): align evidence and review guards with explicit routes * fix(workflows): clarify ship recovery and functional QA evidence * fix(workflows): clarify approval recovery and full QA coverage * refactor(workflows): order review transactions and clarify ship state * fix(ship): clarify final verification and fail closed at publication * fix(evals): attribute native atomic documentation writes * fix(ship): clarify recovery and documentation lifecycle guidance * fix(test): preserve observed native placeholder styling in CI * fix(codex): report watchdog timeouts without a process-exit race * Checkpoint functional QA implementation and workflow validation repairs * Fix documentation and shared-review fixture contracts * docs: clarify judge reuse and evaluation supervision * test: align review evidence and selected case contracts * test: verify append-only documentation checkpoints and recovery * fix: qualify QA workflows and CI validation repairs * fix: launch shared-libs fixture scripts on Windows * fix: qualify QA deadlines, fixture isolation, and shard cleanup * fix: preserve qualified QA and cancellation repairs * fix: enforce functional fixture authority and share strict event decoding * fix: retain free-test evidence and explain recovery * fix: reject malformed native evidence after decoder consolidation * test: use reliable capture for telemetry privacy filters * test: refresh measured quick coverage and document validation costs * Fix native fixture receipts and preserve VM validation evidence * Align negative judge controls with upstream clarity policy * Fix report-only QA preparation and public evidence handling * Clarify QA-only preparation and current-report preservation * Stream Ship quality judgments with an explicit 64k response contract * Validate compact judge reasoning locally with supported wire schema * Align functional QA fixture instructions with evidence acceptance * Bind native browser diagnostics to execution evidence and align review verdicts * Preserve native diagnostic line boundaries * Serialize functional QA evidence from native captures * Keep large QA evidence fixture payload out of Windows argv
418 lines
30 KiB
TypeScript
418 lines
30 KiB
TypeScript
import { expect, test } from 'bun:test';
|
|
import { qaFunctionalPrompt, QA_FUNCTIONAL_CASES } from './helpers/qa-functional-eval';
|
|
import { qaCommandAllowed } from './helpers/qa-functional-observer';
|
|
import { mkdtempSync, readFileSync, realpathSync, rmSync, writeFileSync } from 'node:fs';
|
|
import { basename, dirname, join } from 'node:path';
|
|
import { tmpdir } from 'node:os';
|
|
import { parseNDJSON } from './helpers/session-runner';
|
|
import { qaFunctionalVerdict, qaNativeProbes } from './helpers/qa-functional-evidence';
|
|
import { createQAFunctionalFixture, ownedPath } from './helpers/qa-functional-fixture';
|
|
import { validateQACheckpoints } from './helpers/qa-checkpoint-evidence';
|
|
import { computePaidCaseSelection } from '../scripts/test-paid-shards';
|
|
|
|
test.each(['full', 'pr'] as const)('%s selection assigns the captured webhook regression to its native owner', profile => {
|
|
for (const file of ['qa-webhook-r85-checkpoints.json', 'qa-functional-ci-36505065023.json']) {
|
|
const result = computePaidCaseSelection({ profile, env: {},
|
|
changedFiles: [`test/fixtures/${file}`] });
|
|
expect(result.selection).toEqual({ e2e: ['qa-functional-webhook-report'], judges: [] });
|
|
if (profile === 'pr') {
|
|
expect(result.coverage?.mode).toBe('pr');
|
|
expect(result.coverage?.unknownFiles).toEqual([]);
|
|
}
|
|
}
|
|
});
|
|
|
|
test.each(['full', 'pr'] as const)('%s selection assigns the captured CLI learning regression to its native owner', profile => {
|
|
const result = computePaidCaseSelection({ profile, env: {},
|
|
changedFiles: ['test/fixtures/qa-functional-cli-learning-ci-36516246523.json'] });
|
|
expect(result.selection).toEqual({ e2e: ['qa-functional-cli-report'], judges: [] });
|
|
});
|
|
|
|
test('CI CLI replay summaries fail only the distinct-probe metric despite valid native exploration', () => {
|
|
const captures = JSON.parse(readFileSync(join(import.meta.dir, 'fixtures/qa-functional-cli-learning-ci-36516246523.json'), 'utf8'));
|
|
expect(captures.attempts).toHaveLength(2);
|
|
for (const original of captures.attempts) {
|
|
const fixture = createQAFunctionalFixture('cli');
|
|
try {
|
|
const captured = JSON.parse(JSON.stringify(original).replaceAll(original.fixtureRoot, fixture.root));
|
|
fixture.revision = captured.report.revision;
|
|
captured.report.runtime = `bun ${Bun.version}`;
|
|
for (const [name, content] of Object.entries(captured.reports)) writeFileSync(ownedPath(fixture.root, `qa-reports/${name}`), content as string);
|
|
const parsed = parseNDJSON(captured.publicEvents.map(event => JSON.stringify(event)));
|
|
const result = { ...parsed, output: '', exitReason: captured.exitReason, browseErrors: [], duration: 0,
|
|
firstResponseMs: 0, maxInterTurnMs: 0, model: 'native-event-replay',
|
|
costEstimate: { inputChars: 0, outputChars: 0, estimatedTokens: 0, estimatedCost: 0, turnsUsed: 0 } };
|
|
const read = parsed.toolCalls.find(call => call.tool === 'Read' && call.input.file_path.endsWith('/qa/sections/system-functional.md'))!;
|
|
const section = { path: 'qa/sections/system-functional.md', content: read.output.replace(/^\s*\d+(?:→|\t)/gm, '').trim() };
|
|
const verdict = (report = captured.report, transcript = captured.publicEvents, observation = captured.observation) =>
|
|
qaFunctionalVerdict(fixture, 'qa-only', { ...result, transcript }, observation, report, section, captured.reports['report.md']);
|
|
expect(verdict()).toEqual(['missing observation-to-next-hypothesis evidence']);
|
|
expect(captured.report.learning[0].observationCommand).toBe(captured.report.learning[0].nextCommand);
|
|
const noteCall = parsed.toolCalls.find(call => call.tool === 'Write' && basename(call.input.file_path).startsWith('exploration-')
|
|
&& JSON.parse(call.input.content).observationCommand !== JSON.parse(call.input.content).nextCommand)!;
|
|
const { observationCommand, nextCommand, hypothesis } = JSON.parse(noteCall.input.content);
|
|
const learning = { observationCommand, nextCommand, hypothesis };
|
|
const report = { ...captured.report, learning: [learning] };
|
|
expect(verdict(report)).toEqual([]);
|
|
for (const invalid of [[], [{ ...learning, nextCommand: observationCommand }],
|
|
[{ ...learning, nextCommand: 'bun run probe -- apply uncaptured 11' }],
|
|
[{ ...learning, nextCommand: 'bun test' }],
|
|
[{ ...learning, nextCommand: `${observationCommand}; ${nextCommand}` }],
|
|
[{ ...learning, observationCommand: nextCommand, nextCommand: observationCommand }],
|
|
[{ ...learning, hypothesis: 'Try another probe.' }]]) {
|
|
expect(verdict({ ...report, learning: invalid })).toContain('missing observation-to-next-hypothesis evidence');
|
|
}
|
|
const noteId = captured.publicEvents.flatMap(event => event.message.content)
|
|
.find(block => block.type === 'tool_use' && block.name === 'Write' && block.input.file_path === noteCall.input.file_path).id;
|
|
const pending = captured.publicEvents.filter(event => !event.message.content.some(block => block.type === 'tool_result' && block.tool_use_id === noteId));
|
|
expect(pending.length).toBeLessThan(captured.publicEvents.length);
|
|
expect(verdict(report, pending).some(failure => failure.includes('checkpoint'))).toBe(true);
|
|
expect(verdict(report, captured.publicEvents, { ...captured.observation, complete: false })).toContain('incomplete write observation');
|
|
const noteFile = join(fixture.root, 'qa-reports', basename(noteCall.input.file_path));
|
|
const note = JSON.parse(readFileSync(noteFile, 'utf8'));
|
|
note.observed.stdout = 'invented output';
|
|
writeFileSync(noteFile, JSON.stringify(note));
|
|
expect(verdict(report).some(failure => failure.includes('checkpoint'))).toBe(true);
|
|
} finally { fixture.cleanup(); }
|
|
}
|
|
});
|
|
|
|
test('functional driver discloses its learning, CLI coverage and repair acceptance requirements', () => {
|
|
for (const entry of QA_FUNCTIONAL_CASES) {
|
|
const prompt = qaFunctionalPrompt(entry);
|
|
expect(prompt).toContain(`Read ${entry.mode}/SKILL.md, qa/sections/scope.md, ${entry.mode}/sections/exploratory.md and qa/sections/system-functional.md in full`);
|
|
expect(prompt).toContain('All four reads are required before probing in this fixture, even when its surfaces and isolation are already established');
|
|
expect(prompt).toContain('command is the exact full outer capture invocation, including that ID and all wrapper options, not just the native child command after --');
|
|
expect(prompt).toContain('"command":"<exact full outer capture invocation>"');
|
|
expect(prompt).not.toContain('<exact executed native probe command>');
|
|
expect(prompt).toContain('different later command');
|
|
expect(prompt).toContain('not the required same-command replay');
|
|
expect(prompt).toContain('Select that checkpoint ID in annotations.learning');
|
|
expect(prompt).toContain('the production helper copies its observationCommand, hypothesis and nextCommand');
|
|
expect(prompt).toContain('different native child commands');
|
|
expect(prompt).toContain('Only the helper writes observed fields');
|
|
expect(prompt).toContain('English, more than 20 characters');
|
|
if (entry.family === 'cli') expect(prompt).toContain('a successful apply; balance alone is not enough');
|
|
if (entry.mode === 'qa') {
|
|
expect(prompt).toContain(`repair only src/${entry.family === 'cli' ? 'cli' : 'worker'}.ts`);
|
|
expect(prompt).toContain('existing tests remain read-only');
|
|
expect(prompt).toContain('Freeze all test files after red');
|
|
}
|
|
}
|
|
});
|
|
|
|
test('CI native checkpoint keeps public fixture paths exact and requires the completed Write', () => {
|
|
const captured = JSON.parse(readFileSync(join(import.meta.dir, 'fixtures/qa-functional-ci-36505065023.json'), 'utf8'));
|
|
const reportRoot = realpathSync(mkdtempSync(join(tmpdir(), 'qa-ci-note-')));
|
|
try {
|
|
for (const variant of ['captured redaction', 'exact public JSON', 'omitted path', 'renamed identity', 'pending Write', 'late Write']) {
|
|
const originalRoot = dirname(captured.checkpointEvents[2].message.content[0].input.file_path);
|
|
const events = JSON.parse(JSON.stringify(captured.checkpointEvents).replaceAll(originalRoot, reportRoot));
|
|
const parsed = parseNDJSON(events.map(event => JSON.stringify(event)));
|
|
const probes = qaNativeProbes(parsed);
|
|
expect(probes.map(probe => probe.command)).toEqual(['bun run probe -- duplicate', 'bun run probe -- partial']);
|
|
const write = events[2].message.content[0].input;
|
|
const note = JSON.parse(write.content);
|
|
expect(note.observed.stateRoot).toContain('/qa-state-redacted/');
|
|
expect(probes[0].observed.stateRoot).toContain('/qaf-QXv0tB/.qa-state/');
|
|
if (variant !== 'captured redaction') note.observed = structuredClone(probes[0].observed);
|
|
if (variant === 'omitted path') delete note.observed.stateRoot;
|
|
if (variant === 'renamed identity') {
|
|
note.observed.fixture = note.observed.stateRoot;
|
|
delete note.observed.stateRoot;
|
|
}
|
|
write.content = JSON.stringify(note);
|
|
writeFileSync(write.file_path, write.content);
|
|
if (variant === 'pending Write') events.splice(3, 1);
|
|
if (variant === 'late Write') events.push(...events.splice(3, 1));
|
|
const errors = validateQACheckpoints({ transcript: events, reportRoot, probes,
|
|
requiredProbes: probes.slice(1), files: { 'exploration-002.json': write.content },
|
|
reportMarkdown: '[Checkpoint](exploration-002.json)' });
|
|
if (variant === 'exact public JSON') expect(errors).toEqual([]);
|
|
else expect(errors).toContain('QA checkpoint: Missing unique completed checkpoint before probe: bun run probe -- partial');
|
|
}
|
|
} finally { rmSync(reportRoot, { recursive: true, force: true }); }
|
|
});
|
|
|
|
test('CI native exploration Read does not stand in for completed method Reads', () => {
|
|
const captured = JSON.parse(readFileSync(join(import.meta.dir, 'fixtures/qa-functional-ci-36505065023.json'), 'utf8'));
|
|
const fixture = createQAFunctionalFixture('webhook');
|
|
try {
|
|
for (const variant of ['omitted methods', 'pending methods', 'completed methods']) {
|
|
const events = structuredClone(captured.omittedReadEvents);
|
|
if (variant !== 'omitted methods') events.push(...captured.completedMethodEvents.filter(event => variant === 'completed methods' || event.type === 'assistant'));
|
|
const parsed = parseNDJSON(events.map(event => JSON.stringify(event)));
|
|
expect(parsed.toolCalls.some(call => call.tool === 'Read' && call.input.file_path.endsWith('/qa-only/sections/exploratory.md') && call.output.includes('# Shared exploratory QA'))).toBe(true);
|
|
const errors = qaFunctionalVerdict(fixture, 'qa-only', {
|
|
...parsed, output: '', exitReason: 'success', browseErrors: [], duration: 0,
|
|
firstResponseMs: 0, maxInterTurnMs: 0, model: 'native-event-replay',
|
|
costEstimate: { inputChars: 0, outputChars: 0, estimatedTokens: 0, estimatedCost: 0, turnsUsed: 0 },
|
|
}, { complete: true, failures: [], events: [], changed: [], before: {}, after: {}, limits: [] }, {},
|
|
{ path: 'qa/sections/system-functional.md', content: readFileSync(join(import.meta.dir, '../qa/sections/system-functional.md'), 'utf8') });
|
|
expect(errors.includes('no completed functional instruction read')).toBe(variant !== 'completed methods');
|
|
}
|
|
} finally { fixture.cleanup(); }
|
|
});
|
|
|
|
test('the native launcher consumes the family-specific actor boundary', () => {
|
|
const source = readFileSync(join(import.meta.dir, 'helpers/qa-functional-eval.ts'), 'utf8');
|
|
expect(source).toContain('prompt: qaFunctionalPrompt(entry)');
|
|
for (const entry of QA_FUNCTIONAL_CASES) {
|
|
const prompt = qaFunctionalPrompt(entry);
|
|
expect(prompt).toContain(`Read ${entry.mode}/SKILL.md`);
|
|
expect(prompt).toContain(entry.mode === 'qa' ? 'Full exploration and the Standard fix tier' : 'Full report-only exploration');
|
|
expect(prompt).not.toContain('at Standard depth');
|
|
expect(prompt).toContain('successful checkpoint publication before the next probe');
|
|
expect(prompt).toContain('no shell composition, scripts or added path operands');
|
|
expect(prompt).toContain('ONLY complete JSON actually emitted');
|
|
expect(prompt).toContain('never a combined command list');
|
|
expect(prompt).toContain('Put tests, raw CLI diagnostics, launch failures and timeouts in Markdown');
|
|
expect(prompt).toContain(entry.family === 'cli'
|
|
? 'The generic wrapper does NOT support wait'
|
|
: 'bun cancel.ts is a CLI-only entrypoint, not part of this fixture');
|
|
expect(prompt).not.toContain('parseInt');
|
|
if (entry.family === 'webhook') {
|
|
expect(prompt).toContain('All eight scenarios are required coverage; a replay does not replace another scenario');
|
|
expect(prompt).toContain('Choose their order from observations after the happy path');
|
|
} else {
|
|
expect(prompt).not.toContain('All eight scenarios');
|
|
}
|
|
}
|
|
});
|
|
|
|
test('declared examples respect the existing closed native grammar', () => {
|
|
for (const command of ['pwd', 'ls', 'ls -la', 'git status --short', 'git status --porcelain',
|
|
'git branch --show-current', 'git diff', 'git diff --stat', 'git rev-parse HEAD', 'bun --version',
|
|
'date -u +%Y-%m-%dT%H:%M:%SZ', 'bun test', 'bun test test/contract.test.ts',
|
|
'bun run probe -- balance', 'bun run probe -- export', 'bun run probe -- apply id 7',
|
|
'bun run probe -- apply', 'bun cancel.ts', ...['happy', 'reject', 'duplicate', 'partial',
|
|
'concurrent-ab', 'concurrent-ba', 'cancel', 'dependency'].map(name => `bun run probe -- ${name}`)]) {
|
|
expect(qaCommandAllowed(command)).toBe(true);
|
|
}
|
|
for (const command of ['ls -la .qa-state qa-reports', 'ls -la .qa-state', 'ls -la qa-reports', 'bun run probe -- wait bad 7',
|
|
'git rev-parse HEAD; bun --version', 'bun run probe -- happy && bun run probe -- partial']) {
|
|
expect(qaCommandAllowed(command)).toBe(false);
|
|
}
|
|
});
|
|
|
|
test('artifact completion preserves exact evidence before concise linked reporting', () => {
|
|
for (const entry of QA_FUNCTIONAL_CASES) {
|
|
const prompt = qaFunctionalPrompt(entry);
|
|
expect(prompt).toContain('Materialize qa-reports/evidence.json first, then write a concise qa-reports/report.md');
|
|
expect(prompt).toContain('using the functional report structure');
|
|
expect(prompt).toContain('Link the evidence and checkpoint files rather than repeating full probe payloads in Markdown');
|
|
expect(prompt).toContain('Both artifacts are required before completion');
|
|
expect(prompt).toContain('The learning array is a summary: choose one completed checkpoint');
|
|
expect(prompt).toContain('not another probe or a duplicate of the complete checkpoint ledger');
|
|
expect(prompt).toContain('Both commands must name exact captured probes with different native child commands');
|
|
expect(prompt).toContain('Preserve every checkpoint and link every checkpoint in Markdown');
|
|
expect(prompt).toContain('keep every executed probe and its complete JSON in evidence');
|
|
expect(prompt).toContain('Evidence rows contain ONLY complete JSON actually emitted by native probes, including failures and repeats');
|
|
expect(prompt).toContain('retain pre-repair results alongside green results');
|
|
expect(prompt).toContain('Never synthesize JSON');
|
|
expect(prompt).toContain('one causal sentence per checkpoint hypothesis (English, more than 20 characters) and compact JSON formatting, preserving every field and value');
|
|
expect(prompt).toContain('retain its headings and required fields');
|
|
expect(prompt).toContain('link to evidence.json and checkpoints for details already recorded there');
|
|
expect(prompt).toContain('After saving both artifacts, return only their paths and the actual completion status');
|
|
expect(prompt).toContain('Never shorten native JSON or omit a required probe, check or field');
|
|
expect(prompt).toContain('aim under 400 words');
|
|
}
|
|
const source = readFileSync(join(import.meta.dir, 'helpers/qa-functional-eval.ts'), 'utf8');
|
|
expect(source).toContain('maxTurns: 40');
|
|
expect(source).toContain('completionReserveMs: timeout / 4');
|
|
});
|
|
|
|
test('fix completion budgets for required repair and avoids duplicating preserved evidence', () => {
|
|
for (const entry of QA_FUNCTIONAL_CASES) {
|
|
const prompt = qaFunctionalPrompt(entry);
|
|
if (entry.mode === 'qa') {
|
|
expect(prompt).toContain('a reproduced in-tier defect requires the authorized native regression, repair and verification');
|
|
expect(prompt).toContain('retain its headings and required fields');
|
|
expect(prompt).toContain('link to evidence.json and checkpoints for details already recorded there');
|
|
expect(prompt).toContain('Include the diagnosis, red/green test results and coverage limits');
|
|
expect(prompt).toContain('After saving both artifacts, return only their paths and the actual completion status');
|
|
expect(prompt).toContain('Never shorten native JSON or omit a required probe, check or field');
|
|
const stages = ['1. Prove the regression red', '2. On the repaired source', '3. Save the evidence and Markdown artifacts'];
|
|
const positions = stages.map(stage => prompt.indexOf(stage));
|
|
expect(positions.every(position => position >= 0)).toBe(true);
|
|
expect(positions).toEqual([...positions].sort((a, b) => a - b));
|
|
expect(prompt).toContain('A green test suite does not substitute for these native probes');
|
|
expect(prompt).toContain('not a signal to stop stage 2');
|
|
expect(prompt).toContain('report incomplete; do not call it complete with a caveat');
|
|
expect(prompt).toContain('one causal sentence per checkpoint hypothesis (English, more than 20 characters) and compact JSON formatting, preserving every field and value');
|
|
} else {
|
|
expect(prompt).not.toContain('This is a fix run');
|
|
expect(prompt).toContain('Include the diagnosis, proposed test stubs and coverage limits');
|
|
expect(prompt).not.toContain('Include the diagnosis, red/green test results');
|
|
}
|
|
}
|
|
});
|
|
|
|
test('webhook fix stage retains the required scenarios omitted by both R29 captures', () => {
|
|
const captured = [
|
|
{ id: 'ecd6da06-abd0-4299-8c33-e1b99a672325', scenarios: ['happy', 'partial', 'partial', 'concurrent-ab', 'partial', 'cancel', 'dependency', 'happy'], missing: ['reject', 'duplicate', 'concurrent-ba'] },
|
|
{ id: '45722f13-a72c-4c01-87cc-8e17285ef8c4', scenarios: ['happy', 'concurrent-ab', 'concurrent-ab', 'concurrent-ab', 'happy', 'cancel', 'dependency'], missing: ['reject', 'duplicate', 'partial', 'concurrent-ba'] },
|
|
];
|
|
const prompt = qaFunctionalPrompt({ family: 'webhook', mode: 'qa' });
|
|
const verification = prompt.slice(prompt.indexOf('2. On the repaired source'), prompt.indexOf('3. Save the evidence'));
|
|
const required = ['happy', 'reject', 'duplicate', 'partial', 'concurrent-ab', 'concurrent-ba', 'cancel', 'dependency'];
|
|
for (const attempt of captured) {
|
|
expect(required.filter(scenario => !attempt.scenarios.includes(scenario))).toEqual(attempt.missing);
|
|
for (const scenario of required) expect(verification).toContain(`\`${scenario}\``);
|
|
}
|
|
expect(verification).toContain('every still-unobserved scenario');
|
|
expect(verification).toContain('recheck earlier scenarios affected by the repair');
|
|
expect(verification).toContain('None of these scenarios is optional exploration');
|
|
expect(prompt).toContain('the completion reserve does not end required coverage');
|
|
expect(qaFunctionalPrompt({ family: 'cli', mode: 'qa' })).not.toContain('`concurrent-ba`');
|
|
});
|
|
|
|
test('fix-stage checkpoint provenance survives intervening native regression tests', () => {
|
|
const prompt = qaFunctionalPrompt({ family: 'webhook', mode: 'qa' });
|
|
expect(prompt).toContain('most recent completed native probe');
|
|
expect(prompt).toContain('Tests, source edits and clock reads do not replace that observation');
|
|
expect(prompt).toContain('put red/green test output in the report, not in observed');
|
|
expect(prompt).toContain('write no checkpoint when there is no next probe');
|
|
});
|
|
|
|
test('R29 captured webhook bytes bind across a green test; test summaries and altered JSON do not', () => {
|
|
const captured = {
|
|
before: '{"scenario":"concurrent-ab","requests":[{"method":"POST","path":"/events","auth":"$QA_SYNTHETIC_AUTH","body":{"id":"delivery","cents":7},"status":202,"response":"{\\"accepted\\":\\"delivery\\"}"}],"order":["a","b"],"interrupted":"","state":{"jobs":{"delivery":{"cents":7,"status":"complete","attempts":2}},"effects":[{"id":"delivery","cents":7},{"id":"delivery","cents":7}]},"stateRoot":"/q/gstack-paid-shard-hyrPGY/tmp/qaf-JaNRb2/.qa-state/concurrent-ab-nVrq2v"}',
|
|
after: '{"scenario":"concurrent-ab","requests":[{"method":"POST","path":"/events","auth":"$QA_SYNTHETIC_AUTH","body":{"id":"delivery","cents":7},"status":202,"response":"{\\"accepted\\":\\"delivery\\"}"}],"order":["a","b"],"interrupted":"","state":{"jobs":{"delivery":{"cents":7,"status":"complete","attempts":1}},"effects":[{"id":"delivery","cents":7}]},"stateRoot":"/q/gstack-paid-shard-hyrPGY/tmp/qaf-JaNRb2/.qa-state/concurrent-ab-ySExJy"}',
|
|
green: 'bun test v1.4.0 (34cbb9a40)\n\n 2 pass\n 0 fail\n 4 expect() calls\nRan 2 tests across 2 files. [50.00ms]',
|
|
checkpoint: '{"observationCommand":"bun test test/worker.regression-1.test.ts","observed":"red before repair: expect(received).toEqual(expected) — effects had two {id:delivery,cents:7} entries; 0 pass 1 fail. After the src/worker.ts post-gate recheck, bun test reported 2 pass 0 fail (native test output, not probe JSON).","hypothesis":"The regression turned green after the post-gate ledger recheck, so the original failing native probe should now show exactly one effect with both workers still released in a then b order.","nextCommand":"bun run probe -- concurrent-ab"}\n',
|
|
};
|
|
const reportRoot = realpathSync(mkdtempSync(join(tmpdir(), 'qa-r29-')));
|
|
const name = 'exploration-004.json';
|
|
const file = join(reportRoot, name);
|
|
const command = 'bun run probe -- concurrent-ab';
|
|
try {
|
|
for (const variant of ['original summary', 'native JSON', 'raw test output', 'missing stateRoot', 'green result', 'missing Write receipt', 'terminal note']) {
|
|
const note = JSON.parse(captured.checkpoint);
|
|
if (variant !== 'original summary') {
|
|
note.observationCommand = command;
|
|
note.observed = JSON.parse(captured.before);
|
|
}
|
|
if (variant === 'raw test output') { note.observationCommand = 'bun test'; note.observed = captured.green; }
|
|
if (variant === 'missing stateRoot') delete note.observed.stateRoot;
|
|
if (variant === 'green result') note.observed = JSON.parse(captured.after);
|
|
if (variant === 'terminal note') note.nextCommand = 'none';
|
|
const content = variant === 'original summary' ? captured.checkpoint : JSON.stringify(note);
|
|
writeFileSync(file, content, { mode: 0o600 });
|
|
const calls = [
|
|
{ tool: 'Bash', input: { command }, output: `$ bun probe.ts concurrent-ab\n${captured.before}` },
|
|
{ tool: 'Bash', input: { command: 'bun test' }, output: captured.green },
|
|
{ tool: 'Write', input: { file_path: file, content }, output: `File created successfully at: ${file}` },
|
|
{ tool: 'Bash', input: { command }, output: `$ bun probe.ts concurrent-ab\n${captured.after}` },
|
|
];
|
|
const packets = calls.flatMap((call, index) => [
|
|
{ type: 'assistant', message: { content: [{ type: 'tool_use', id: `r29-${index}`, name: call.tool, input: call.input }] } },
|
|
...variant === 'missing Write receipt' && call.tool === 'Write' ? [] : [
|
|
{ type: 'user', message: { content: [{ type: 'tool_result', tool_use_id: `r29-${index}`, content: call.output }] } },
|
|
],
|
|
]);
|
|
const result = parseNDJSON(packets.map(packet => JSON.stringify(packet)));
|
|
const probes = qaNativeProbes(result);
|
|
expect(probes).toHaveLength(2);
|
|
const failures = validateQACheckpoints({ transcript: result.transcript, reportRoot, probes,
|
|
requiredProbes: probes.slice(1), files: { [name]: content }, reportMarkdown: `[Checkpoint](${name})` });
|
|
if (variant === 'native JSON') expect(failures).toEqual([]);
|
|
else expect(failures).toContain(`QA checkpoint: Missing unique completed checkpoint before probe: ${command}`);
|
|
if (variant === 'original summary') expect(failures).toContain(`QA checkpoint: Unrelated, reused or retrospective checkpoint: ${name}`);
|
|
}
|
|
} finally { rmSync(reportRoot, { recursive: true, force: true }); }
|
|
});
|
|
|
|
test.each(JSON.parse(readFileSync(join(import.meta.dir, 'fixtures/qa-webhook-r85-checkpoints.json'), 'utf8')))(
|
|
'R85 $attempt rejects a published draft even after a corrected successor', capture => {
|
|
for (const variant of ['captured pair', 'complete note only', 'draft only', 'missing Write receipt', 'missing report link', 'partial observation']) {
|
|
const reportRoot = realpathSync(mkdtempSync(join(tmpdir(), 'qa-r85-')));
|
|
try {
|
|
const packets = structuredClone(capture.transcript);
|
|
if (variant === 'draft only') packets.splice(4, 2);
|
|
else if (variant !== 'captured pair') packets.splice(2, 2);
|
|
if (variant === 'missing Write receipt') packets.splice(3, 1);
|
|
const files: Record<string, string> = {};
|
|
for (const packet of packets) {
|
|
for (const block of packet.message.content) {
|
|
if (block.type !== 'tool_use' || block.name !== 'Write') continue;
|
|
const name = basename(block.input.file_path);
|
|
block.input.file_path = join(reportRoot, name);
|
|
if (variant === 'partial observation') {
|
|
const note = JSON.parse(block.input.content);
|
|
delete note.observed.state;
|
|
block.input.content = JSON.stringify(note);
|
|
}
|
|
files[name] = block.input.content;
|
|
writeFileSync(block.input.file_path, block.input.content);
|
|
}
|
|
}
|
|
const result = parseNDJSON(packets.map((packet: unknown) => JSON.stringify(packet)));
|
|
const probes = qaNativeProbes(result);
|
|
expect(probes).toHaveLength(2);
|
|
const failures = validateQACheckpoints({ transcript: result.transcript, reportRoot, probes,
|
|
requiredProbes: probes.slice(1), files,
|
|
reportMarkdown: variant === 'missing report link' ? '' : Object.keys(files).map(name => `[Checkpoint](${name})`).join('\n') });
|
|
if (variant === 'complete note only') expect(failures).toEqual([]);
|
|
else if (variant === 'captured pair' || variant === 'draft only') {
|
|
expect(failures).toContain(`QA checkpoint: ${capture.bad === 'exploration-003.json' ? 'Invalid checkpoint schema' : 'Unrelated, reused or retrospective checkpoint'}: ${capture.bad}`);
|
|
} else if (variant === 'missing report link') {
|
|
expect(failures).toContain(`QA checkpoint: Report does not link checkpoint: ${capture.good}`);
|
|
} else {
|
|
expect(failures).toContain(`QA checkpoint: Missing unique completed checkpoint before probe: ${probes[1].command}`);
|
|
}
|
|
} finally { rmSync(reportRoot, { recursive: true, force: true }); }
|
|
}
|
|
},
|
|
);
|
|
|
|
test('report-only exploration requires a completed written checkpoint before the next probe', () => {
|
|
const section = readFileSync(join(import.meta.dir, '../qa-only/sections/exploratory.md'), 'utf8');
|
|
const positions = ['1. First demonstrate success', '2. **Decide whether another probe is needed.**', '**Publish before probing.**', '3. Run that exact probe; G enforces the deadline when bounded']
|
|
.map(marker => section.indexOf(marker));
|
|
expect(positions.every(position => position >= 0)).toBe(true);
|
|
expect(positions).toEqual([...positions].sort((a, b) => a - b));
|
|
expect(section).toContain('exploration-NNN.json');
|
|
expect(section).toContain("Reuse resolved REPORT_DIR");
|
|
expect(section).toContain('own a fresh');
|
|
expect(section).toContain('owned probe directory');
|
|
for (const field of ['observationCommand', 'observed', 'hypothesis', 'nextCommand']) expect(section).toContain(`${field}:`);
|
|
expect(section).toContain('Wait for successful checkpoint publication');
|
|
expect(section).toContain('Never backfill or overwrite notes');
|
|
expect(section).toContain('link each checkpoint');
|
|
expect(section).not.toContain('a separate assistant text message');
|
|
});
|
|
|
|
test('surface evidence checks defer to one exploratory execution sequence', () => {
|
|
const source = readFileSync(join(import.meta.dir, '../scripts/resolvers/qa.ts'), 'utf8');
|
|
expect(source).toContain('Each probe is one native command/interaction plus checks, excluding bookkeeping');
|
|
const positions = ['2. **Decide whether another probe is needed.**', '**Publish before probing.**', '3. Run that exact probe; G enforces the deadline when bounded']
|
|
.map(marker => source.indexOf(marker));
|
|
expect(positions.every(position => position >= 0)).toBe(true);
|
|
expect(positions).toEqual([...positions].sort((a, b) => a - b));
|
|
expect(source).toContain('Never batch probes');
|
|
expect(source).toContain('Follow the shared exploratory loop\'s order and written checkpoints');
|
|
expect(source).toContain('Replay the exact failing command/request from the same initial fixture state');
|
|
expect(source).toContain('Another input or a regression test is not that replay');
|
|
});
|
|
|
|
test('all public callers directly require the functional method before exploration', () => {
|
|
for (const skill of ['qa', 'qa-only', 'review', 'ship']) {
|
|
const file = skill === 'ship' ? 'ship/sections/review-army.md' : `${skill}/SKILL.md`;
|
|
const source = readFileSync(join(import.meta.dir, '..', file), 'utf8');
|
|
expect(source).toContain(['review', 'ship'].includes(skill) ? '../qa/sections/exploratory.md' : 'sections/exploratory.md');
|
|
expect(source).not.toMatch(/Functional surfaces[^\n]*\n[^\n]*Read[^\n]*system-functional\.md/);
|
|
const explorer = readFileSync(join(import.meta.dir, '..', skill === 'qa-only' ? 'qa-only' : 'qa', 'sections/exploratory.md'), 'utf8');
|
|
expect(explorer).toMatch(/Functional surfaces[^\n]*\n[^\n]*Read[^\n]*system-functional\.md/);
|
|
expect(explorer).toContain('Browser surfaces only');
|
|
const stages = ['Read `sections/scope.md`', 'in full and select the surfaces',
|
|
'Read `sections/system-functional.md`', 'Write a **charter**', '1. First demonstrate success'];
|
|
const positions = stages.map(stage => explorer.indexOf(stage));
|
|
expect(positions.every(position => position >= 0)).toBe(true);
|
|
expect(positions).toEqual([...positions].sort((a, b) => a - b));
|
|
const functional = readFileSync(join(import.meta.dir, '../qa/sections/system-functional.md'), 'utf8');
|
|
expect(functional).toContain('## Contract map');
|
|
expect(functional).toContain("Follow the shared exploratory loop's order and written checkpoints");
|
|
}
|
|
});
|