mirror of
https://github.com/garrytan/gstack.git
synced 2026-10-04 10:26:52 +02:00
feat(qa-evidence): materialize computes the phase verdict; callers must report it
Approved by Garry: the helper, not the model, decides whether evidence can
pass. materialize writes verdict {status, open} into evidence.json and prints
it: fail or blocked from row classifications, inconclusive while any row is
superseded, a complete capture is withheld, a declared required probe is
unrun or there is no evidence, else pass. The caller fixture requires
receipt.status to equal that verdict. CI late-input kept reporting pass with a
superseded happy probe.
This commit is contained in:
1 parent
1137cba7ac
commit
06545fcc30
3 files changed
+27
-5
No files matched your search
+16
-3
@@ -294,10 +294,23 @@ function materialize(root: string, source: string) {
|
||||
}
|
||||
return { observationCommand: note.observationCommand, hypothesis: note.hypothesis, nextCommand: note.nextCommand };
|
||||
});
|
||||
const sha256 = publish(root, 'evidence.json', { ...annotations, evidence, learning });
|
||||
return { action: 'materialize', status: 'complete', sha256, annotationsSha256: hash(bytes), exitCode: 0,
|
||||
const classes = annotations.evidence.map((row: any) => String(row.classification).toLowerCase());
|
||||
const open = [
|
||||
...annotations.evidence.filter((row: any) => String(row.classification).toLowerCase() === 'superseded').map((row: any) => `capture ${row.capture} superseded`),
|
||||
...completeCaptures(root).filter(capture => !captures.has(capture)).map(capture => `capture ${capture} withheld`),
|
||||
...(requiredRemaining(root).requiredRemaining ?? []).map(command => `required probe not run: ${command}`),
|
||||
...(annotations.evidence.length ? [] : ['no evidence rows']),
|
||||
];
|
||||
const verdict = {
|
||||
status: classes.some((value: string) => /fail|defect/.test(value)) ? 'fail'
|
||||
: classes.some((value: string) => /block/.test(value)) ? 'blocked'
|
||||
: open.length || classes.some((value: string) => !['pass', 'superseded'].includes(value)) ? 'inconclusive' : 'pass',
|
||||
open,
|
||||
};
|
||||
const sha256 = publish(root, 'evidence.json', { ...annotations, evidence, learning, verdict });
|
||||
return { action: 'materialize', status: 'complete', sha256, annotationsSha256: hash(bytes), exitCode: 0, verdict,
|
||||
reportLinks: notes.map(note => `[checkpoint ${note.name.slice(12, 15)}](${note.name})`),
|
||||
next: 'Include every reportLinks entry in the Markdown report.' };
|
||||
next: `Include every reportLinks entry in the Markdown report, and report the overall status as ${verdict.status}${verdict.open.length ? ` (open: ${verdict.open.join('; ')})` : ''}; rerun what is open first if a pass is required.` };
|
||||
}
|
||||
|
||||
const QA_EVIDENCE_USAGE = 'capture ROOT ID [--public] --deadline FILE|--timeout-ms MS -- COMMAND ARGS | checkpoint ROOT ID CAPTURE OBSERVATION_COMMAND HYPOTHESIS NEXT_COMMAND | checkpoint ROOT ID INTENT_FILE | materialize ROOT ANNOTATIONS (annotations: {evidence: [{capture, command, contract, expected, classification}], limits: [..]}; revision, runtime, cwd and learning are filled in)';
|
||||
|
||||
@@ -444,6 +444,12 @@ export function validateCallerEvidence(input: {
|
||||
errors.push('review completion preceded handoff freshness decision');
|
||||
}
|
||||
}
|
||||
if (input.requireCapturedEvidence && input.reportRoot) {
|
||||
const evidenceFile = path.join(input.reportRoot, 'evidence.json');
|
||||
const verdict = fs.existsSync(evidenceFile) ? JSON.parse(fs.readFileSync(evidenceFile, 'utf8')).verdict : undefined;
|
||||
if (!verdict) errors.push('missing helper verdict in evidence.json');
|
||||
else if (input.receipt.status !== verdict.status) errors.push(`receipt status ${input.receipt.status} differs from helper verdict ${verdict.status}`);
|
||||
}
|
||||
if (input.receipt.status === 'pass' && (input.receipt.remaining.length || selected.some(probe => probe?.status !== 'pass'))) {
|
||||
errors.push('blocked, failing or incomplete coverage reported green');
|
||||
}
|
||||
@@ -673,7 +679,7 @@ export function callerReviewRecordTemplate(fixture: Pick<QaCallerFixture, 'calle
|
||||
|
||||
export function qaCallerSessionOptions(fixture: QaCallerFixture, runId: string): Parameters<typeof runSkillTest>[0] {
|
||||
return {
|
||||
prompt: `Load gstack's /${fixture.caller} supplied parent phase from caller-${fixture.caller}.md and resume it on the selected working-tree diff against origin/main. This excerpt comes from ${fixture.runtime}/${fixture.caller}/SKILL.md; resolve installed-relative references there, not from the excerpt file or product directory. That path identifies the asset base, not another entrypoint: do not read or invoke the full parent SKILL.md or rerun its preamble. Earlier preamble/branch/base setup is complete; use the existing local origin/main ref without fetch. Earlier-phase asset locators are ${fixture.runtime}/review/checklist.md and ${fixture.runtime}/qa/templates/functional-report-template.md. Read those files directly when referenced; recursive Glob does not follow the installed asset symlinks. Cross-project learnings are configured off in this owned fixture. ${fixture.reviewStart ? `The actual review-start helper already captured REVIEW_START=${fixture.reviewStart} for this unchanged core pass; retain that token. ` : ''} This fixture evaluates only the supplied parent phase, not later publication stages. The fixture owner recorded these read-only observations when this phase began; they are current until an input changes, so use them instead of re-running those commands, and recheck freshness before completion outputs:\n${fixture.startObservations}\n Read README.md for the project contract and commands. Use diagnostic-client commands such as \`bun scripts/probe.ts <literal>\` for exploratory discoveries and their checkpoint evidence. A required \`bun run test\` is separate suite verification: report it as verification, never as a diagnostic observation or checkpoint anchor/target. Use the production evidence helper to publish each diagnostic checkpoint as \`reports/exploration-NNN.json\`, not inside a nested directory; do not transcribe its observed payload. ${fixture.caseId === 'ship-exploratory-plan-checks' ? 'The previously discovered plan is PLAN.md.' : 'No plan file was found.'} There is no remote service and no release publication is authorized. There is no interactive approver; do not invent answers or permission. Keep normal parent decision gates. Before the completion report and each completed:true review record, read HANDOFF.md and reports/HANDOFF.md if present for any concurrent collaborator update, await the results, and compare evidence with current inputs. If a probe's snapshot differs from an earlier probe's, read reports/HANDOFF.md first; it names any collaborator change. A gstack-review-log completed:true record is a completion, not preliminary bookkeeping; a later handoff read cannot validate an earlier completion.\n\nDeadline bookkeeping additionally permits \`bun ${fixture.runtime}/bin/gstack-qa-deadline start ${fixture.cwd}/reports/deadline.json SECONDS [EARLIER_UTC]\`, \`bun ${fixture.runtime}/bin/gstack-qa-deadline status ${fixture.cwd}/reports/deadline.json\`, and \`bun ${fixture.runtime}/bin/gstack-qa-deadline run ${fixture.cwd}/reports/deadline.json -- bun scripts/probe.ts [literal]\`. These are closed literal forms: SECONDS must be positive and at most 300, EARLIER_UTC is the optional caller absolute deadline: use the section clock's Hard deadline UTC, never its Runner entry UTC, reserve-start time or a clock-read time. The child is only the existing diagnostic client with zero or one literal argument. Resolve these exact helper and state paths; do not use variables, another helper, another state file, nested wrappers, scripts, operators or substitutions. Only this helper may create or change reports/deadline.json and its .qa-deadline- temporary files; never use Write/Edit/MultiEdit on those paths. Record the full outer run command in checkpoints and evidence; keep the unchanged child JSON as observed, separate from prefixed guard diagnostics. A completed expired guard-run is not a probe or a pass: retain its unused checkpoint, report not-run coverage and do not restart the deadline. Keep the 12-probe smoke limit. Required suites and explicit plan checks are outside the bounded smoke budget, not permission to reset it.\n\nFunctional evidence uses the same production helper and existing diagnostic client: \`bun ${fixture.runtime}/bin/gstack-qa-evidence capture ${fixture.cwd}/reports NNN --public --deadline ${fixture.cwd}/reports/deadline.json -- bun scripts/probe.ts [literal]\`. These diagnostic receipts are declared public/synthetic, so --public is approved; a fresh three-digit ID is required each time. Explicit plan probes outside the smoke budget may replace --deadline with --timeout-ms 10000; this does not reset or bypass the smoke deadline. Publish causal intent with \`bun ${fixture.runtime}/bin/gstack-qa-evidence checkpoint ${fixture.cwd}/reports NNN CAPTURE_ID 'full prior capture command' 'causal hypothesis' 'full next capture command'\`; quote arguments literally. For complex quoting, Write only capture, observationCommand, hypothesis and nextCommand to reports/intent
Line truncated
|
||||
prompt: `Load gstack's /${fixture.caller} supplied parent phase from caller-${fixture.caller}.md and resume it on the selected working-tree diff against origin/main. This excerpt comes from ${fixture.runtime}/${fixture.caller}/SKILL.md; resolve installed-relative references there, not from the excerpt file or product directory. That path identifies the asset base, not another entrypoint: do not read or invoke the full parent SKILL.md or rerun its preamble. Earlier preamble/branch/base setup is complete; use the existing local origin/main ref without fetch. Earlier-phase asset locators are ${fixture.runtime}/review/checklist.md and ${fixture.runtime}/qa/templates/functional-report-template.md. Read those files directly when referenced; recursive Glob does not follow the installed asset symlinks. Cross-project learnings are configured off in this owned fixture. ${fixture.reviewStart ? `The actual review-start helper already captured REVIEW_START=${fixture.reviewStart} for this unchanged core pass; retain that token. ` : ''} This fixture evaluates only the supplied parent phase, not later publication stages. The fixture owner recorded these read-only observations when this phase began; they are current until an input changes, so use them instead of re-running those commands, and recheck freshness before completion outputs:\n${fixture.startObservations}\n Read README.md for the project contract and commands. Use diagnostic-client commands such as \`bun scripts/probe.ts <literal>\` for exploratory discoveries and their checkpoint evidence. A required \`bun run test\` is separate suite verification: report it as verification, never as a diagnostic observation or checkpoint anchor/target. Use the production evidence helper to publish each diagnostic checkpoint as \`reports/exploration-NNN.json\`, not inside a nested directory; do not transcribe its observed payload. ${fixture.caseId === 'ship-exploratory-plan-checks' ? 'The previously discovered plan is PLAN.md.' : 'No plan file was found.'} There is no remote service and no release publication is authorized. There is no interactive approver; do not invent answers or permission. Keep normal parent decision gates. Before the completion report and each completed:true review record, read HANDOFF.md and reports/HANDOFF.md if present for any concurrent collaborator update, await the results, and compare evidence with current inputs. If a probe's snapshot differs from an earlier probe's, read reports/HANDOFF.md first; it names any collaborator change. A gstack-review-log completed:true record is a completion, not preliminary bookkeeping; a later handoff read cannot validate an earlier completion.\n\nDeadline bookkeeping additionally permits \`bun ${fixture.runtime}/bin/gstack-qa-deadline start ${fixture.cwd}/reports/deadline.json SECONDS [EARLIER_UTC]\`, \`bun ${fixture.runtime}/bin/gstack-qa-deadline status ${fixture.cwd}/reports/deadline.json\`, and \`bun ${fixture.runtime}/bin/gstack-qa-deadline run ${fixture.cwd}/reports/deadline.json -- bun scripts/probe.ts [literal]\`. These are closed literal forms: SECONDS must be positive and at most 300, EARLIER_UTC is the optional caller absolute deadline: use the section clock's Hard deadline UTC, never its Runner entry UTC, reserve-start time or a clock-read time. The child is only the existing diagnostic client with zero or one literal argument. Resolve these exact helper and state paths; do not use variables, another helper, another state file, nested wrappers, scripts, operators or substitutions. Only this helper may create or change reports/deadline.json and its .qa-deadline- temporary files; never use Write/Edit/MultiEdit on those paths. Record the full outer run command in checkpoints and evidence; keep the unchanged child JSON as observed, separate from prefixed guard diagnostics. A completed expired guard-run is not a probe or a pass: retain its unused checkpoint, report not-run coverage and do not restart the deadline. Keep the 12-probe smoke limit. Required suites and explicit plan checks are outside the bounded smoke budget, not permission to reset it.\n\nFunctional evidence uses the same production helper and existing diagnostic client: \`bun ${fixture.runtime}/bin/gstack-qa-evidence capture ${fixture.cwd}/reports NNN --public --deadline ${fixture.cwd}/reports/deadline.json -- bun scripts/probe.ts [literal]\`. These diagnostic receipts are declared public/synthetic, so --public is approved; a fresh three-digit ID is required each time. Explicit plan probes outside the smoke budget may replace --deadline with --timeout-ms 10000; this does not reset or bypass the smoke deadline. Publish causal intent with \`bun ${fixture.runtime}/bin/gstack-qa-evidence checkpoint ${fixture.cwd}/reports NNN CAPTURE_ID 'full prior capture command' 'causal hypothesis' 'full next capture command'\`; quote arguments literally. For complex quoting, Write only capture, observationCommand, hypothesis and nextCommand to reports/intent
Line truncated
|
||||
appendSystemPrompt: `Caller execution scheduling (fixture contract):
|
||||
This session has at most 25 assistant turns, including required verification and final artifacts. The command boundary applies to each Bash call, not to the number of independent tool calls in an assistant turn.
|
||||
After required clock and approval prerequisites settle, issue independent source Reads and read-only discovery together as separate native tool calls once their paths and inputs are known. For example, after the first clock read, one response can Read the caller excerpt, README.md, every product file and the referenced checklist and templates. Wait for their results before decisions that depend on them.
|
||||
|
||||
@@ -331,7 +331,10 @@ test('materialize refuses evidence whose declared input snapshot predates the la
|
||||
expect(stale.status).toBe(2);
|
||||
expect(receipt(stale.stderr).message).toContain('capture 001 observed an older input snapshot than the latest capture 002');
|
||||
f.json('annotations.json', { revision: 'fixture-revision', limits: ['Adverse coverage predates the input change.'], evidence: [{ ...rows[0], classification: 'superseded' }, rows[1]] });
|
||||
expect(f.run('materialize', f.root, 'annotations.json').status).toBe(0);
|
||||
const materialized = f.run('materialize', f.root, 'annotations.json');
|
||||
expect(materialized.status).toBe(0);
|
||||
expect(receipt(materialized.stdout).verdict).toEqual({ status: 'inconclusive', open: ['capture 001 superseded'] });
|
||||
expect(JSON.parse(fs.readFileSync(path.join(f.root, 'evidence.json'), 'utf8')).verdict.status).toBe('inconclusive');
|
||||
});
|
||||
|
||||
test('captures list declared-but-unrun required probes without judging them', () => {
|
||||
|
||||
Reference in new issue
Block a user