mirror of
https://github.com/garrytan/gstack.git
synced 2026-10-03 01:46:55 +02:00
v1.91.9.0 feat: test value bar in plan-eng-review, review, qa and ship, plus /test-audit (#2998)
This commit is contained in:
1 parent
943105f109
commit
96764e80a6
56 files changed
+2445
-216
No files matched your search
@@ -0,0 +1,156 @@
|
||||
/** Test value bar behavior in /ship Step 7, the /review testing specialist and
|
||||
* /test-audit. Each case is one bounded capture on a small fixture:
|
||||
* ship-coverage-value a ★-only path lands in weak_gaps and X <= Y
|
||||
* review-test-value a source-grep test and a test-only export are
|
||||
* INFORMATIONAL findings; the SKILL.md golden is not
|
||||
* test-audit-report-only the same fixture yields complete retirement cards,
|
||||
* keeps the golden and edits nothing
|
||||
*/
|
||||
import { afterAll } from 'bun:test';
|
||||
import * as fs from 'node:fs';
|
||||
import * as path from 'node:path';
|
||||
import * as os from 'node:os';
|
||||
import { spawnSync } from 'node:child_process';
|
||||
import { CAPTURE_MS } from './helpers/eval-budgets';
|
||||
import { runSkillTest } from './helpers/session-runner';
|
||||
import { ROOT, runId, describeIfSelected, testIfSelected, copyDirSync, logCost,
|
||||
createEvalCollector, finalizeEvalCollector } from './helpers/e2e-helpers';
|
||||
import { extractSkillBody } from './helpers/skill-fixture';
|
||||
import { createCoverageAuditFixture } from './fixtures/coverage-audit-fixture';
|
||||
import { createTestValueFixture, lastJsonLine, jsonFindings, LOW_VALUE_TESTS } from './helpers/test-value-fixture';
|
||||
import { runRecordedOfficeHoursAttempt, OFFICE_HOURS_BUN_GRACE_MS } from './helpers/office-hours-attempt';
|
||||
import { resolveEvalModel } from '../lib/eval-model';
|
||||
import { RETIREMENT_FIELDS } from '../scripts/resolvers/test-value';
|
||||
import type { SkillTestResult } from './helpers/session-runner';
|
||||
|
||||
const evalCollector = createEvalCollector('e2e');
|
||||
const RUNNER_MS = CAPTURE_MS - 3 * OFFICE_HOURS_BUN_GRACE_MS;
|
||||
|
||||
function trackedChanges(cwd: string): string {
|
||||
return spawnSync('git', ['status', '--porcelain', '--untracked-files=no'], { cwd, encoding: 'utf8', timeout: 10_000 }).stdout.trim();
|
||||
}
|
||||
|
||||
const SMOKE_TEST = `
|
||||
import { refundPayment } from '../src/billing';
|
||||
test('refundPayment does not throw', () => {
|
||||
expect(() => refundPayment('pay_1', 'duplicate')).not.toThrow();
|
||||
});
|
||||
`;
|
||||
|
||||
type Case = {
|
||||
id: string;
|
||||
skill: string;
|
||||
suite: string;
|
||||
setup: (cwd: string) => void;
|
||||
prompt: (cwd: string, reportDir: string) => string;
|
||||
validate: (result: SkillTestResult, cwd: string, reportDir: string) => void;
|
||||
};
|
||||
|
||||
const CASES: Case[] = [
|
||||
{
|
||||
id: 'ship-coverage-value', skill: 'ship', suite: 'Ship Test Value E2E',
|
||||
setup: cwd => {
|
||||
createCoverageAuditFixture(cwd);
|
||||
fs.appendFileSync(path.join(cwd, 'test/billing.test.ts'), SMOKE_TEST);
|
||||
},
|
||||
prompt: cwd => `Read ship/SKILL.md and ship/sections/test-coverage.md for the current ship workflow.
|
||||
|
||||
You are on the feature/billing branch. The base branch is main. There is no remote.
|
||||
Run ONLY the Step 7 coverage audit, inline, as the audit subagent would, applying it
|
||||
directly to ${cwd}/src/billing.ts and its tests in ${cwd}/test/billing.test.ts (a
|
||||
targeted audit with no branch diff). Generation: audit-only; passes used: 0 of 2.
|
||||
Do not dispatch subagents, write tests or modify files. Skip every other step.
|
||||
End with the audit's LAST-line JSON exactly as Step 7 specifies.`,
|
||||
validate: result => {
|
||||
const json = lastJsonLine(result.output || '');
|
||||
if (!json) throw new Error('ship coverage: no last-line JSON');
|
||||
if (typeof json.coverage_pct !== 'number' || typeof json.coverage_pct_value !== 'number') throw new Error(`ship coverage: numeric coverage_pct and coverage_pct_value required, got ${JSON.stringify(json)}`);
|
||||
if (json.coverage_pct < json.coverage_pct_value) throw new Error(`ship coverage: coverage_pct ${json.coverage_pct} < coverage_pct_value ${json.coverage_pct_value}`);
|
||||
const weak = Array.isArray(json.weak_gaps) ? json.weak_gaps : [];
|
||||
if (!weak.some((gap: any) => /refund/i.test(JSON.stringify(gap)))) throw new Error(`ship coverage: the ★-only refundPayment path must be in weak_gaps, got ${JSON.stringify(json.weak_gaps)}`);
|
||||
},
|
||||
},
|
||||
{
|
||||
id: 'review-test-value', skill: 'review', suite: 'Review Test Value E2E',
|
||||
setup: cwd => createTestValueFixture(cwd),
|
||||
prompt: () => `Read review/SKILL.md and review/sections/review-army.md for the current review workflow.
|
||||
Apply ONLY the testing specialist checklist in review/specialists/testing.md to this
|
||||
branch's diff (\`git diff main...HEAD\`), as the testing specialist would. Do not
|
||||
dispatch other specialists, fix anything or modify files. Output the specialist's JSON
|
||||
findings, one per line.`,
|
||||
validate: (result, cwd) => {
|
||||
const findings = jsonFindings(result.output || '');
|
||||
const about = (needle: RegExp) => findings.filter(finding => needle.test(`${finding.path} ${finding.summary}`));
|
||||
const grep = about(/pricing-source/);
|
||||
const seam = about(new RegExp(`pricing-reset|${LOW_VALUE_TESTS.testOnlySymbol}`));
|
||||
if (!grep.length || !grep.every(finding => finding.severity === 'INFORMATIONAL')) throw new Error(`review: source-grep test needs an INFORMATIONAL finding, got ${JSON.stringify(grep)}`);
|
||||
if (!seam.length || !seam.every(finding => finding.severity === 'INFORMATIONAL')) throw new Error(`review: test-only export needs an INFORMATIONAL finding, got ${JSON.stringify(seam)}`);
|
||||
if (!seam.some(finding => /non_test_callers/.test(JSON.stringify(finding.evidence ?? '')) && /git grep/.test(JSON.stringify(finding.evidence ?? '')))) throw new Error('review: test-only export evidence must record non_test_callers and the git grep search');
|
||||
const golden = about(/skill-golden/);
|
||||
if (golden.length) throw new Error(`review: the SKILL.md golden test must not be flagged, got ${JSON.stringify(golden)}`);
|
||||
if (trackedChanges(cwd)) throw new Error('review: tracked files changed');
|
||||
},
|
||||
},
|
||||
{
|
||||
id: 'test-audit-report-only', skill: 'test-audit', suite: 'Test Audit Report-Only E2E',
|
||||
setup: cwd => createTestValueFixture(cwd),
|
||||
prompt: (_cwd, reportDir) => `Read test-audit/SKILL.md and run /test-audit on this repository, report-only.
|
||||
Treat this as a headless session: ask no questions, approve no batch, edit no file in
|
||||
the repository. There is no gstack install here, so skip the SLUG setup line and write
|
||||
the report to ${reportDir}/test-audit.md and its JSON sidecar to
|
||||
${reportDir}/test-audit.json instead. Stop after Step 4.`,
|
||||
validate: (_result, cwd, reportDir) => {
|
||||
const sidecarPath = path.join(reportDir, 'test-audit.json');
|
||||
if (!fs.existsSync(path.join(reportDir, 'test-audit.md'))) throw new Error('test-audit: report missing');
|
||||
if (!fs.existsSync(sidecarPath)) throw new Error('test-audit: JSON sidecar missing');
|
||||
const sidecar = JSON.parse(fs.readFileSync(sidecarPath, 'utf8'));
|
||||
const candidates: any[] = Array.isArray(sidecar.candidates) ? sidecar.candidates : [];
|
||||
const retiring = candidates.filter(candidate => candidate.verdict !== 'retain');
|
||||
for (const needle of [/pricing-source/, new RegExp(`pricing-reset|${LOW_VALUE_TESTS.testOnlySymbol}`)]) {
|
||||
const match = retiring.find(candidate => needle.test(String(candidate.test)));
|
||||
if (!match) throw new Error(`test-audit: missing candidate ${needle}, got ${JSON.stringify(candidates.map(c => c.test))}`);
|
||||
const missing = RETIREMENT_FIELDS.filter(field => !String(match.retirement_card?.[field] ?? '').trim());
|
||||
if (missing.length) throw new Error(`test-audit: ${match.test} retirement card missing ${missing.join(', ')}`);
|
||||
}
|
||||
if (retiring.some(candidate => /skill-golden/.test(String(candidate.test)))) throw new Error('test-audit: the SKILL.md golden must be retained');
|
||||
if (trackedChanges(cwd)) throw new Error('test-audit: tracked files changed');
|
||||
},
|
||||
},
|
||||
];
|
||||
|
||||
for (const entry of CASES) describeIfSelected(entry.suite, [entry.id], () => {
|
||||
testIfSelected(entry.id, async () => {
|
||||
let cwd: string | undefined;
|
||||
let reportDir: string | undefined;
|
||||
try {
|
||||
await runRecordedOfficeHoursAttempt({
|
||||
collector: evalCollector, name: entry.id, suite: entry.suite,
|
||||
model: process.env.EVALS_MODEL ?? resolveEvalModel('capture'),
|
||||
budgetMs: CAPTURE_MS - OFFICE_HOURS_BUN_GRACE_MS,
|
||||
run: async signal => {
|
||||
cwd = fs.mkdtempSync(path.join(os.tmpdir(), `skill-e2e-${entry.id}-`));
|
||||
reportDir = fs.mkdtempSync(path.join(os.tmpdir(), `skill-e2e-${entry.id}-report-`));
|
||||
entry.setup(cwd);
|
||||
copyDirSync(path.join(ROOT, entry.skill), path.join(cwd, entry.skill));
|
||||
fs.writeFileSync(path.join(cwd, entry.skill, 'SKILL.md'), extractSkillBody(path.join(ROOT, entry.skill)));
|
||||
fs.writeFileSync(path.join(cwd, '.git', 'info', 'exclude'), `${entry.skill}/\n`);
|
||||
return runSkillTest({
|
||||
prompt: entry.prompt(cwd, reportDir),
|
||||
workingDirectory: cwd, maxTurns: 25,
|
||||
allowedTools: ['Bash', 'Read', 'Write', 'Glob', 'Grep'],
|
||||
timeout: RUNNER_MS, testName: entry.id, runId, signal,
|
||||
});
|
||||
},
|
||||
validate: result => {
|
||||
logCost(entry.id, result);
|
||||
if (result.exitReason !== 'success') throw new Error(`${entry.id}: ${result.exitReason}`);
|
||||
entry.validate(result, cwd!, reportDir!);
|
||||
},
|
||||
});
|
||||
} finally {
|
||||
for (const dir of [cwd, reportDir]) if (dir) try { fs.rmSync(dir, { recursive: true, force: true }); } catch {}
|
||||
}
|
||||
}, CAPTURE_MS);
|
||||
});
|
||||
|
||||
afterAll(async () => { await finalizeEvalCollector(evalCollector); });
|
||||
Reference in new issue
Block a user