mirror of
https://github.com/garrytan/gstack.git
synced 2026-10-04 10:26:52 +02:00
fix(review): pass Review Army checklists by path, run research alongside dispatch, always probe the design detector; state review-log invocation and statuses in the caller fixture
- review-army-perf-n-plus-one: the parent copied full checklists into agent prompts and ran web research before dispatch (290 s on a 12-line diff); 212 s now. - review-design-lite: 5 of 6 captured trials reported the detector absent without probing; the probe is mandatory and its first line is reported, and the contract credits only fake-engine rule ids the checklist never names. - review-exploratory-small-cli: the fixture never gave review-log's direct invocation or status vocabulary; the model ran it through bun and wrote status "blocked". The prompt states both and the validator rejects out-of-vocabulary review statuses. Each case passed a focused paid run after repair.
This commit is contained in:
1 parent
a18e6cf655
commit
a06d22e52a
15 files changed
+155
-62
No files matched your search
@@ -18,3 +18,14 @@ export function installFakeImpeccable(prefix = 'gstack-fake-impeccable-'): { dir
|
||||
fs.copyFileSync(DETECT_SAMPLE, path.join(dir, 'impeccable-detect-sample.json')); // the shim's documented default output, beside it
|
||||
return { dir, bin };
|
||||
}
|
||||
|
||||
/** Sample rule ids the design checklist never names: a review can carry them only from the detector's rows. */
|
||||
export function detectorOnlyRuleIds(checklist: string): string[] {
|
||||
const rules = JSON.parse(fs.readFileSync(DETECT_SAMPLE, 'utf-8')) as Array<{ antipattern: string }>;
|
||||
return [...new Set(rules.map(rule => rule.antipattern))].filter(id => !checklist.includes(id));
|
||||
}
|
||||
|
||||
export function carriesDetectorRows(review: string, checklist: string): boolean {
|
||||
const text = review.toLowerCase();
|
||||
return detectorOnlyRuleIds(checklist).some(id => new RegExp(`(?<![\\w-])${id}(?![\\w-])`).test(text));
|
||||
}
|
||||
@@ -248,6 +248,8 @@ export function qaCallerCommandAllowed(command: string, workflowCommands: string
|
||||
|| /^\/?[\w./-]+\/bin\/gstack-(?:review-read|specialist-stats)$/.test(text);
|
||||
}
|
||||
|
||||
const REVIEW_RECORD_STATUSES: Record<QaCaller, string[]> = { review: ['clean', 'issues_found'], ship: ['clean', 'issues_found', 'unavailable'] };
|
||||
|
||||
export function validateCallerEvidence(input: {
|
||||
caller: QaCaller;
|
||||
result: Pick<SkillTestResult, 'transcript' | 'exitReason'>;
|
||||
@@ -300,6 +302,11 @@ export function validateCallerEvidence(input: {
|
||||
if (tool.name === 'Bash' && !qaCallerCommandAllowed(String(tool.input.command), input.workflowCommands, deadline)) {
|
||||
errors.push('command outside declared caller observation interface');
|
||||
}
|
||||
const record = tool.name === 'Bash' ? String(tool.input.command).match(/gstack-review-log '(.*)'(?: --finish \S+)?$/s)?.[1] : undefined;
|
||||
const status = record?.match(/"status"\s*:\s*"([^"]*)"/)?.[1] ?? '';
|
||||
if (record && /"skill"\s*:\s*"review"/.test(record) && !REVIEW_RECORD_STATUSES[input.caller].includes(status)) {
|
||||
errors.push(`review record status outside the ${input.caller} vocabulary: ${status}`);
|
||||
}
|
||||
if (tool.name === 'Read' && /\/(?:browse|devex-review)\/SKILL\.md$|\/qa\/sections\/(?:browser-[^/]+|qa-patterns)\.md$/.test(file)) {
|
||||
errors.push(`unexpected browser/DX load: ${file}`);
|
||||
}
|
||||
@@ -633,7 +640,7 @@ process.exit(exit ?? 127);
|
||||
|
||||
export function qaCallerSessionOptions(fixture: QaCallerFixture, runId: string): Parameters<typeof runSkillTest>[0] {
|
||||
return {
|
||||
prompt: `Load gstack's /${fixture.caller} supplied parent phase from caller-${fixture.caller}.md and resume it on the selected working-tree diff against origin/main. This excerpt comes from ${fixture.runtime}/${fixture.caller}/SKILL.md; resolve installed-relative references there, not from the excerpt file or product directory. That path identifies the asset base, not another entrypoint: do not read or invoke the full parent SKILL.md or rerun its preamble. Earlier preamble/branch/base setup is complete; use the existing local origin/main ref without fetch. Earlier-phase asset locators are ${fixture.runtime}/review/checklist.md and ${fixture.runtime}/qa/templates/functional-report-template.md. Read those files directly when referenced; recursive Glob does not follow the installed asset symlinks. Cross-project learnings are configured off in this owned fixture. ${fixture.reviewStart ? `The actual review-start helper already captured REVIEW_START=${fixture.reviewStart} for this unchanged core pass; retain that token. ` : ''} This fixture evaluates only the supplied parent phase, not later publication stages. Read README.md for the project contract and commands. Use diagnostic-client commands such as \`bun scripts/probe.ts <literal>\` for exploratory discoveries and their checkpoint evidence. A required \`bun run test\` is separate suite verification: report it as verification, never as a diagnostic observation or checkpoint anchor/target. Use the production evidence helper to publish each diagnostic checkpoint as \`reports/exploration-NNN.json\`, not inside a nested directory; do not transcribe its observed payload. ${fixture.caseId === 'ship-exploratory-plan-checks' ? 'The previously discovered plan is PLAN.md.' : 'No plan file was found.'} There is no remote service and no release publication is authorized. There is no interactive approver; do not invent answers or permission. Keep normal parent decision gates. Before every completion report or bookkeeping log, read HANDOFF.md and reports/HANDOFF.md if present for any concurrent collaborator update, await the results, and compare evidence with current inputs. A gstack-review-log completed:true record is a completion, not preliminary bookkeeping; a later handoff read cannot validate an earlier completion.\n\nDeadline bookkeeping additionally permits \`bun ${fixture.runtime}/bin/gstack-qa-deadline start ${fixture.cwd}/reports/deadline.json SECONDS [EARLIER_UTC]\`, \`bun ${fixture.runtime}/bin/gstack-qa-deadline status ${fixture.cwd}/reports/deadline.json\`, and \`bun ${fixture.runtime}/bin/gstack-qa-deadline run ${fixture.cwd}/reports/deadline.json -- bun scripts/probe.ts [literal]\`. These are closed literal forms: SECONDS must be positive and at most 300, EARLIER_UTC is the optional caller absolute deadline: use the section clock's Hard deadline UTC, never its Runner entry UTC, reserve-start time or a clock-read time. The child is only the existing diagnostic client with zero or one literal argument. Resolve these exact helper and state paths; do not use variables, another helper, another state file, nested wrappers, scripts, operators or substitutions. Only this helper may create or change reports/deadline.json and its .qa-deadline- temporary files; never use Write/Edit/MultiEdit on those paths. Record the full outer run command in checkpoints and evidence; keep the unchanged child JSON as observed, separate from prefixed guard diagnostics. A completed expired guard-run is not a probe or a pass: retain its unused checkpoint, report not-run coverage and do not restart the deadline. Keep the 12-probe smoke limit. Required suites and explicit plan checks are outside the bounded smoke budget, not permission to reset it.\n\nFunctional evidence uses the same production helper and existing diagnostic client: \`bun ${fixture.runtime}/bin/gstack-qa-evidence capture ${fixture.cwd}/reports NNN --public --deadline ${fixture.cwd}/reports/deadline.json -- bun scripts/probe.ts [literal]\`. These diagnostic receipts are declared public/synthetic, so --public is approved; a fresh three-digit ID is required each time. Explicit plan probes outside the smoke budget may replace --deadline with --timeout-ms 10000; this does not reset or bypass the smoke deadline. Publish causal intent with \`bun ${fixture.runtime}/bin/gstack-qa-evidence checkpoint ${fixture.cwd}/reports NNN CAPTURE_ID 'full prior capture command' 'causal hypothesis' 'full next capture command'\`; quote arguments literally. For complex quoting, Write only capture, observationCommand, hypothesis and nextCommand to reports/intent.json; publish with the same helper: checkpoint REPORT_ROOT NNN intent.json. Materialize is supported when evidence.json is required. Sources stay inside reports. Decide to execute the next probe before publishing its checkpoint, then await successful publication and dispatch that exact probe. If you defer an optional idea or stop exploration, do not publish a checkpoint for it; descri
Line truncated
|
||||
prompt: `Load gstack's /${fixture.caller} supplied parent phase from caller-${fixture.caller}.md and resume it on the selected working-tree diff against origin/main. This excerpt comes from ${fixture.runtime}/${fixture.caller}/SKILL.md; resolve installed-relative references there, not from the excerpt file or product directory. That path identifies the asset base, not another entrypoint: do not read or invoke the full parent SKILL.md or rerun its preamble. Earlier preamble/branch/base setup is complete; use the existing local origin/main ref without fetch. Earlier-phase asset locators are ${fixture.runtime}/review/checklist.md and ${fixture.runtime}/qa/templates/functional-report-template.md. Read those files directly when referenced; recursive Glob does not follow the installed asset symlinks. Cross-project learnings are configured off in this owned fixture. ${fixture.reviewStart ? `The actual review-start helper already captured REVIEW_START=${fixture.reviewStart} for this unchanged core pass; retain that token. ` : ''} This fixture evaluates only the supplied parent phase, not later publication stages. Read README.md for the project contract and commands. Use diagnostic-client commands such as \`bun scripts/probe.ts <literal>\` for exploratory discoveries and their checkpoint evidence. A required \`bun run test\` is separate suite verification: report it as verification, never as a diagnostic observation or checkpoint anchor/target. Use the production evidence helper to publish each diagnostic checkpoint as \`reports/exploration-NNN.json\`, not inside a nested directory; do not transcribe its observed payload. ${fixture.caseId === 'ship-exploratory-plan-checks' ? 'The previously discovered plan is PLAN.md.' : 'No plan file was found.'} There is no remote service and no release publication is authorized. There is no interactive approver; do not invent answers or permission. Keep normal parent decision gates. Before every completion report or bookkeeping log, read HANDOFF.md and reports/HANDOFF.md if present for any concurrent collaborator update, await the results, and compare evidence with current inputs. A gstack-review-log completed:true record is a completion, not preliminary bookkeeping; a later handoff read cannot validate an earlier completion.\n\nDeadline bookkeeping additionally permits \`bun ${fixture.runtime}/bin/gstack-qa-deadline start ${fixture.cwd}/reports/deadline.json SECONDS [EARLIER_UTC]\`, \`bun ${fixture.runtime}/bin/gstack-qa-deadline status ${fixture.cwd}/reports/deadline.json\`, and \`bun ${fixture.runtime}/bin/gstack-qa-deadline run ${fixture.cwd}/reports/deadline.json -- bun scripts/probe.ts [literal]\`. These are closed literal forms: SECONDS must be positive and at most 300, EARLIER_UTC is the optional caller absolute deadline: use the section clock's Hard deadline UTC, never its Runner entry UTC, reserve-start time or a clock-read time. The child is only the existing diagnostic client with zero or one literal argument. Resolve these exact helper and state paths; do not use variables, another helper, another state file, nested wrappers, scripts, operators or substitutions. Only this helper may create or change reports/deadline.json and its .qa-deadline- temporary files; never use Write/Edit/MultiEdit on those paths. Record the full outer run command in checkpoints and evidence; keep the unchanged child JSON as observed, separate from prefixed guard diagnostics. A completed expired guard-run is not a probe or a pass: retain its unused checkpoint, report not-run coverage and do not restart the deadline. Keep the 12-probe smoke limit. Required suites and explicit plan checks are outside the bounded smoke budget, not permission to reset it.\n\nFunctional evidence uses the same production helper and existing diagnostic client: \`bun ${fixture.runtime}/bin/gstack-qa-evidence capture ${fixture.cwd}/reports NNN --public --deadline ${fixture.cwd}/reports/deadline.json -- bun scripts/probe.ts [literal]\`. These diagnostic receipts are declared public/synthetic, so --public is approved; a fresh three-digit ID is required each time. Explicit plan probes outside the smoke budget may replace --deadline with --timeout-ms 10000; this does not reset or bypass the smoke deadline. Publish causal intent with \`bun ${fixture.runtime}/bin/gstack-qa-evidence checkpoint ${fixture.cwd}/reports NNN CAPTURE_ID 'full prior capture command' 'causal hypothesis' 'full next capture command'\`; quote arguments literally. For complex quoting, Write only capture, observationCommand, hypothesis and nextCommand to reports/intent.json; publish with the same helper: checkpoint REPORT_ROOT NNN intent.json. Materialize is supported when evidence.json is required. Sources stay inside reports. Decide to execute the next probe before publishing its checkpoint, then await successful publication and dispatch that exact probe. If you defer an optional idea or stop exploration, do not publish a checkpoint for it; descri
Line truncated
|
||||
appendSystemPrompt: `Caller execution scheduling (fixture contract):
|
||||
This session has at most 25 assistant turns, including required verification and final artifacts. The command boundary applies to each Bash call, not to the number of independent tool calls in an assistant turn.
|
||||
After required clock and approval prerequisites settle, issue independent source Reads and read-only discovery together as separate native tool calls once their paths and inputs are known. Wait for their results before decisions that depend on them.
|
||||
|
||||
Reference in new issue
Block a user