mirror of
https://github.com/garrytan/gstack.git
synced 2026-10-03 01:46:55 +02:00
* feat: add surface-aware exploratory QA and ship documentation gates * test: preserve delegated QA setup authority after main integration * fix(qa): clarify exploration order and preserve report artifacts * test(qa): follow the shared setup reference directly * refactor(ship): make verification and recovery routes explicit * test(ship): align evidence and review guards with explicit routes * fix(workflows): clarify ship recovery and functional QA evidence * fix(workflows): clarify approval recovery and full QA coverage * refactor(workflows): order review transactions and clarify ship state * fix(ship): clarify final verification and fail closed at publication * fix(evals): attribute native atomic documentation writes * fix(ship): clarify recovery and documentation lifecycle guidance * fix(test): preserve observed native placeholder styling in CI * fix(codex): report watchdog timeouts without a process-exit race * Checkpoint functional QA implementation and workflow validation repairs * Fix documentation and shared-review fixture contracts * docs: clarify judge reuse and evaluation supervision * test: align review evidence and selected case contracts * test: verify append-only documentation checkpoints and recovery * fix: qualify QA workflows and CI validation repairs * fix: launch shared-libs fixture scripts on Windows * fix: qualify QA deadlines, fixture isolation, and shard cleanup * fix: preserve qualified QA and cancellation repairs * fix: enforce functional fixture authority and share strict event decoding * fix: retain free-test evidence and explain recovery * fix: reject malformed native evidence after decoder consolidation * test: use reliable capture for telemetry privacy filters * test: refresh measured quick coverage and document validation costs * Fix native fixture receipts and preserve VM validation evidence * Align negative judge controls with upstream clarity policy * Fix report-only QA preparation and public evidence handling * Clarify QA-only preparation and current-report preservation * Stream Ship quality judgments with an explicit 64k response contract * Validate compact judge reasoning locally with supported wire schema * Align functional QA fixture instructions with evidence acceptance * Bind native browser diagnostics to execution evidence and align review verdicts * Preserve native diagnostic line boundaries * Serialize functional QA evidence from native captures * Keep large QA evidence fixture payload out of Windows argv
218 lines
12 KiB
TypeScript
218 lines
12 KiB
TypeScript
import { describe, test, expect } from 'bun:test';
|
||
import * as fs from 'fs';
|
||
import * as path from 'path';
|
||
import { generateReviewDashboard } from '../scripts/resolvers/review';
|
||
import { HOST_PATHS } from '../scripts/resolvers/types';
|
||
import { ALL_HOST_CONFIGS } from '../hosts';
|
||
|
||
/**
|
||
* Template-drift tripwire for the content-binding wave. The bins are
|
||
* code-enforced; the GRADING rules live as prose in rendered templates that
|
||
* agents follow. This test pins the load-bearing rule text in the GENERATED
|
||
* files so a template refactor can't silently drop a rule while the bins keep
|
||
* working. (Prompt-followed prose is honest tier-2 enforcement — this tripwire
|
||
* is what keeps it from being tier-3 vibes.)
|
||
*/
|
||
|
||
const ROOT = path.resolve(import.meta.dir, '..');
|
||
|
||
function rendered(rel: string): string {
|
||
return fs.readFileSync(path.join(ROOT, rel), 'utf-8');
|
||
}
|
||
|
||
describe('content-binding template drift', () => {
|
||
test('design-lite records outside coverage after the outside step in ship', () => {
|
||
const text = rendered('ship/sections/review-army.md');
|
||
const outside = text.indexOf('design voice**');
|
||
expect(outside).toBeGreaterThan(-1);
|
||
expect(text.indexOf('--finish DESIGN_START')).toBeGreaterThan(outside);
|
||
expect(text).toContain('Use the original DESIGN_START token');
|
||
});
|
||
|
||
test('ship eval selection scopes the Rails example below the project-native path', () => {
|
||
const text = rendered('ship/sections/tests.md');
|
||
const native = text.indexOf('**Project-native path:**');
|
||
const rails = text.indexOf('**Rails example only');
|
||
expect(native).toBeGreaterThan(-1);
|
||
expect(rails).toBeGreaterThan(native);
|
||
expect(text).not.toContain('**If no matches:**');
|
||
expect(text).toContain('If any eval fails');
|
||
});
|
||
|
||
test('ship historical readiness does not replace the current pre-landing gate', () => {
|
||
const text = rendered('ship/SKILL.md');
|
||
expect(text).not.toContain('The only review that gates shipping');
|
||
expect(text).toContain('This verdict never skips Step 9 or its finding, approval and convergence gates');
|
||
});
|
||
|
||
test('ship Step 16 carries the evidence check (mechanized IRON LAW)', () => {
|
||
const ship = rendered('ship/SKILL.md');
|
||
expect(ship).toMatch(/gstack-evidence check --label tests --expect-cmd '[^']+' --label vitest --expect-cmd '[^']+' --max-age 24 --allow-paths CHANGELOG\.md,VERSION,package\.json/);
|
||
expect(ship.replace(/\s+/g, ' ')).toContain("| STALE/MISSING: changed content, command or age, or no proven run | Run `~/.claude/skills/gstack/bin/gstack-evidence run --label <lane> -- '<command>'`, read the result and recheck once");
|
||
expect(ship.replace(/\s+/g, ' ')).toContain("**New, changed or unwaived test failure:** STOP publication. Run Steps 5–15, starting with Step 5's triage, then return to Step 16 stage 1");
|
||
expect(ship).toContain('return to Step 16 stage 1');
|
||
});
|
||
|
||
test('ship Step 5 lanes run wrapped with per-lane labels', () => {
|
||
const tests = rendered('ship/sections/tests.md');
|
||
expect(tests).toContain('gstack-evidence run --label tests');
|
||
expect(tests).toContain('gstack-evidence run --label vitest');
|
||
});
|
||
|
||
test('land-and-deploy grades staleness content-first (wtree rule) and checks evidence', () => {
|
||
// Carved (prompt-token-load-reduction): Step 3.5 moved out of the skeleton
|
||
// into the on-demand readiness-gate section — the grading rules live there.
|
||
const land = rendered('land-and-deploy/sections/readiness-gate.md');
|
||
expect(land).toContain('wtree');
|
||
expect(land).toContain('---WTREE---');
|
||
expect(land).toContain('gstack-evidence check --label tests --expect-cmd "$TEST_COMMAND" --max-age 24');
|
||
expect(land).toContain('gstack-evidence run --label tests -- "$TEST_COMMAND"');
|
||
expect(land).toContain('UNKNOWN');
|
||
});
|
||
|
||
test('the review dashboard staleness rule is wtree-first for diff-scoped rows', () => {
|
||
// The dashboard text is generated into every skill that embeds
|
||
// {{REVIEW_DASHBOARD}}; ship is the canonical carrier.
|
||
const ship = rendered('ship/SKILL.md');
|
||
expect(ship).toContain('---WTREE---');
|
||
expect(ship).toContain('Content-first rule');
|
||
expect(ship).toContain('A failed command means UNKNOWN, treated as stale');
|
||
});
|
||
|
||
test('the diff-scoped row list is IDENTICAL in both grading surfaces (no drift)', () => {
|
||
// The resolver (dashboard) and land-and-deploy each carry the row list;
|
||
// they diverged once (codex-review present in one, missing in the other).
|
||
// Rendered dashboards escape backticks (template-literal origin), so match
|
||
// structurally: the three row names in order inside the rule sentence.
|
||
const rowList = /Content-first rule[\s\S]{0,80}?`review`[\s\S]{0,80}?`adversarial-review`[\s\S]{0,80}?`codex-review`[\s\S]{0,80}?ship-stage (?:entries|reviews)[\s\S]{0,80}?`design-review-lite`/;
|
||
expect(rendered('ship/SKILL.md')).toMatch(rowList);
|
||
// land-and-deploy's copy of the row list lives in the carved readiness-gate
|
||
// section (Step 3.5a), not the skeleton.
|
||
expect(rendered('land-and-deploy/sections/readiness-gate.md')).toMatch(rowList);
|
||
});
|
||
|
||
test('both grading surfaces reject missing capture instead of falling back to HEAD', () => {
|
||
for (const file of ['ship/SKILL.md', 'land-and-deploy/sections/readiness-gate.md']) {
|
||
const text = rendered(file);
|
||
expect(text).toContain('review_freshness');
|
||
expect(text).toContain('UNVERIFIED');
|
||
expect(text).toContain('Never fall back');
|
||
expect(text).toMatch(/(?:0|zero) commits/);
|
||
expect(text.toLowerCase()).toMatch(/plan-tier|plan records/);
|
||
}
|
||
});
|
||
|
||
test('diff callers capture before reading and consume the original token', () => {
|
||
const review = rendered('review/SKILL.md');
|
||
expect(review).toContain('gstack-review-log --start review\ngit diff "$DIFF_BASE"');
|
||
expect(review).toContain('--finish REVIEW_START');
|
||
expect(review).toContain('"completed":COMPLETED,"converged":CONVERGED,"cycles":CYCLES');
|
||
const army = rendered('ship/sections/review-army.md');
|
||
expect(army.indexOf('gstack-review-log --start review')).toBeLessThan(army.indexOf('run `git diff origin/<base>`'));
|
||
expect(army).toContain('--finish REVIEW_START');
|
||
expect(army.replace(/\s+/g, ' ')).toContain('Complete items 5–6 exactly once with the original REVIEW_START');
|
||
expect(army).toContain('fixes also require `converged:false`');
|
||
const ship = rendered('ship/SKILL.md');
|
||
expect(army.replace(/\s+/g, ' ')).toContain('**Third fixing cycle reached (`CYCLES >= 3`):** STOP and report recurring findings with `converged:false`; do not run a fourth fixing cycle');
|
||
expect(ship.replace(/\s+/g, ' ')).toContain('Keep the same attempt counts throughout the invocation');
|
||
expect(ship.replace(/\s+/g, ' ')).toContain('a repair never resets approvals or expands them');
|
||
expect(army).toContain('--start design-review-lite');
|
||
expect(army).toContain('--finish DESIGN_START');
|
||
const codex = rendered('codex/sections/review-mode.md');
|
||
const starts = [...codex.matchAll(/gstack-review-log --start codex-review/g)];
|
||
expect(starts).toHaveLength(2);
|
||
expect(starts[0].index).toBeLessThan(codex.indexOf('_gstack_codex_timeout_wrapper 330 codex review'));
|
||
expect(starts[1].index).toBeLessThan(codex.indexOf('git diff "<base>...HEAD"'));
|
||
expect(codex).toContain('--finish CODEX_REVIEW_START');
|
||
expect(codex).toContain('"completed":COMPLETED,"converged":CONVERGED');
|
||
expect(codex).toContain('Fixes stay stale until a genuine rerun');
|
||
for (const skill of ['ship', 'review']) {
|
||
const adversarial = rendered(`${skill}/sections/adversarial.md`);
|
||
expect(adversarial).toContain('--start adversarial-review');
|
||
expect(adversarial).toContain('--finish PASS_START');
|
||
expect(adversarial.replace(/\s+/g, ' ')).toContain('Do the same before each outside adversarial or structured pass reads its diff');
|
||
expect(adversarial).toContain('Each token is consumed once');
|
||
}
|
||
});
|
||
|
||
test('dashboard selection and freshness precede a verdict without replacing the live ship gate', () => {
|
||
for (const host of ALL_HOST_CONFIGS) {
|
||
const text = generateReviewDashboard({ host: host.name, skillName: 'ship',
|
||
tmplPath: 'ship/SKILL.md.tmpl', paths: HOST_PATHS[host.name] }).replace(/\s+/g, ' ');
|
||
const positions = ['**1. Choose the records', '**2. Check freshness',
|
||
'**3. Choose the historical verdict', '**4. Display the dashboard'].map(marker => text.indexOf(marker));
|
||
expect(positions.every(position => position >= 0)).toBe(true);
|
||
expect(positions).toEqual([...positions].sort((a, b) => a - b));
|
||
expect(text).toContain('never substitute an older success for a newer failure');
|
||
expect(text).toContain('CLEARED requires the selected Eng Review to be `clean`, within 7 days and fresh under step 2');
|
||
expect(text).toContain('STALE or UNVERIFIED cannot clear Eng Review');
|
||
expect(text).toContain('Missing `review_freshness`, including legacy log-only records, means UNVERIFIED');
|
||
expect(text).toContain('This verdict never skips Step 9 or its finding, approval and convergence gates');
|
||
expect(text).toContain('Continue Step 1 even when history is NOT CLEARED');
|
||
}
|
||
});
|
||
|
||
test('release-body write side carries the banner tripwire (and it actually fires)', () => {
|
||
const body = rendered('document-release/sections/release-body.md');
|
||
expect(body).toContain('grep -c "UNTRUSTED TRACKER CONTENT" "<run-dir>/body.md"');
|
||
expect(body).toContain('grep -c "UNTRUSTED TRACKER CONTENT" "<run-dir>/body-original.md"');
|
||
// The fail-open shape: grep -c prints 0 AND exits 1 on no-match, so an
|
||
// `|| echo 0` double-emits and breaks the -gt into the clean branch.
|
||
expect(body).not.toContain('|| echo 0');
|
||
expect(body).toContain('banner tripwire clean');
|
||
|
||
// Functional: execute the template's tripwire block against a 0-banner
|
||
// original and a 1-banner outgoing body — the ABORT branch must fire.
|
||
//
|
||
// Pass the script as an ARGV element (spawnSync array form), never by
|
||
// interpolating JSON.stringify into a shell line: JSON escaping is not
|
||
// shell escaping. Inside shell double quotes a JSON "\n" stays a literal
|
||
// backslash-n, which collapsed this multi-line script onto one line where
|
||
// `then\n` became the command word `thenn` and `>&2\nelse\n` became the
|
||
// redirect `>&2nelsen` — silently littering a `2nelsen` file (containing
|
||
// "bash: thenn: command not found") in the repo root on every suite run,
|
||
// while the old not-contains assertion passed vacuously because ALL
|
||
// output had been redirected into that file.
|
||
const block = body.match(/_ORIG_BANNERS=\$\(grep[\s\S]*?fi\n/);
|
||
expect(block).not.toBeNull();
|
||
const fs = require('fs');
|
||
const os = require('os');
|
||
const path = require('path');
|
||
const { spawnSync } = require('child_process');
|
||
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'gstack-banner-'));
|
||
try {
|
||
const scriptFor = (origContent: string, newContent: string) => {
|
||
fs.writeFileSync(path.join(dir, 'orig.md'), origContent);
|
||
fs.writeFileSync(path.join(dir, 'new.md'), newContent);
|
||
return block![0]
|
||
.replaceAll('<run-dir>/body-original.md', path.join(dir, 'orig.md'))
|
||
.replaceAll('<run-dir>/body.md', path.join(dir, 'new.md'));
|
||
};
|
||
|
||
// Banner leaked into the outgoing body → the ABORT branch fires, loudly.
|
||
const abort = spawnSync('bash', ['-c', scriptFor(
|
||
'clean body\n',
|
||
'body with UNTRUSTED TRACKER CONTENT banner leak\n',
|
||
)], { encoding: 'utf-8', timeout: 30_000 });
|
||
expect(abort.stderr).toContain('ABORT: envelope banner leaked');
|
||
expect(abort.stdout).not.toContain('banner tripwire clean');
|
||
|
||
// No banner delta → the clean branch fires.
|
||
const clean = spawnSync('bash', ['-c', scriptFor(
|
||
'clean body\n',
|
||
'also clean body\n',
|
||
)], { encoding: 'utf-8', timeout: 30_000 });
|
||
expect(clean.stdout).toContain('banner tripwire clean');
|
||
expect(clean.stderr).not.toContain('ABORT');
|
||
} finally {
|
||
fs.rmSync(dir, { recursive: true, force: true });
|
||
}
|
||
});
|
||
|
||
test('greptile triage reads bodies through the guard (metadata/body split)', () => {
|
||
const triage = rendered('review/greptile-triage.md');
|
||
expect(triage).toContain('gstack-issue-guard --stdin --source greptile-line');
|
||
expect(triage).toContain('gstack-issue-guard --stdin --source greptile-replies');
|
||
});
|
||
});
|