mirror of
https://github.com/garrytan/gstack.git
synced 2026-10-02 17:40:02 +02:00
4942 lines
250 KiB
TypeScript
4942 lines
250 KiB
TypeScript
import { describe, test, expect, afterAll } from 'bun:test';
|
||
import { assertSinglePreamble } from '../scripts/gen-skill-docs';
|
||
import { validateSkillFrontmatter } from '../scripts/skill-check';
|
||
import { externalHostPathLeaks } from './helpers/skill-parser';
|
||
import { COMMAND_DESCRIPTIONS } from '../browse/src/commands';
|
||
import { SNAPSHOT_FLAGS } from '../browse/src/snapshot';
|
||
import * as fs from 'fs';
|
||
import * as path from 'path';
|
||
import * as os from 'os';
|
||
import { spawnSync } from 'child_process';
|
||
import { runCapturedCommand } from './helpers/sync-command-capture';
|
||
|
||
const ROOT = path.resolve(import.meta.dir, '..');
|
||
const MAX_SKILL_DESCRIPTION_LENGTH = 1024;
|
||
|
||
// Carved-skill aware (v2 plan T9): ship is now a skeleton SKILL.md + sections/*.md.
|
||
// Read the union so assertions about content that MOVED into a section still pass.
|
||
// The skeleton is a subset of the union, so skeleton-only assertions also hold,
|
||
// and negative assertions stay safe (the absent phrases live in neither file).
|
||
function readSkillUnion(skill: string): string {
|
||
let t = fs.readFileSync(path.join(ROOT, skill, 'SKILL.md'), 'utf-8');
|
||
const secDir = path.join(ROOT, skill, 'sections');
|
||
if (fs.existsSync(secDir)) {
|
||
for (const f of fs.readdirSync(secDir).sort()) {
|
||
if (f.endsWith('.md')) t += '\n' + fs.readFileSync(path.join(secDir, f), 'utf-8');
|
||
}
|
||
}
|
||
return t;
|
||
}
|
||
function readShipUnion(): string {
|
||
return readSkillUnion('ship');
|
||
}
|
||
|
||
function readExternalSkillUnion(output: string, hostSubdir: string, skill: string): string {
|
||
const dir = path.join(output, hostSubdir, 'skills', `gstack-${skill}`);
|
||
const sections = path.join(dir, 'sections');
|
||
return fs.readFileSync(path.join(dir, 'SKILL.md'), 'utf-8')
|
||
+ (fs.existsSync(sections) ? fs.readdirSync(sections).sort()
|
||
.filter(file => file.endsWith('.md'))
|
||
.map(file => '\n' + fs.readFileSync(path.join(sections, file), 'utf-8')).join('') : '');
|
||
}
|
||
|
||
|
||
// Token-reduction Phase 1: the preamble's inline bash (session bookkeeping,
|
||
// config echoes, telemetry producers, artifacts sync) moved into
|
||
// bin/gstack-skill-start / bin/gstack-skill-end. The render carries a one-line
|
||
// invocation fence + interpretation prose. Assertions that pinned inline-bash
|
||
// internals now pin the scripts (the new home); render-side assertions pin the
|
||
// fence + prose. Script behavior is pinned by test/gstack-skill-start.test.ts.
|
||
const SKILL_START_SCRIPT = fs.readFileSync(path.join(ROOT, 'bin', 'gstack-skill-start'), 'utf-8');
|
||
const SKILL_END_SCRIPT = fs.readFileSync(path.join(ROOT, 'bin', 'gstack-skill-end'), 'utf-8');
|
||
|
||
function extractDescription(content: string): string {
|
||
const fmEnd = content.indexOf('\n---', 4);
|
||
expect(fmEnd).toBeGreaterThan(0);
|
||
const frontmatter = content.slice(4, fmEnd);
|
||
const lines = frontmatter.split('\n');
|
||
let description = '';
|
||
let inDescription = false;
|
||
const descLines: string[] = [];
|
||
|
||
for (const line of lines) {
|
||
if (line.match(/^description:\s*\|?\s*$/)) {
|
||
inDescription = true;
|
||
continue;
|
||
}
|
||
if (line.match(/^description:\s*\S/)) {
|
||
return line.replace(/^description:\s*/, '').trim();
|
||
}
|
||
if (inDescription) {
|
||
if (line === '' || line.match(/^\s/)) {
|
||
descLines.push(line.replace(/^ /, ''));
|
||
} else {
|
||
break;
|
||
}
|
||
}
|
||
}
|
||
|
||
if (descLines.length > 0) {
|
||
description = descLines.join('\n').trim();
|
||
}
|
||
return description;
|
||
}
|
||
|
||
function extractMarkdownSection(content: string, heading: string): string {
|
||
const escaped = heading.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
||
const startMatch = content.match(new RegExp(`^${escaped}.*$`, 'm'));
|
||
expect(startMatch?.index).toBeDefined();
|
||
const start = startMatch!.index!;
|
||
const afterHeading = start + startMatch![0].length;
|
||
const nextSection = content.slice(afterHeading).match(/\n## /);
|
||
const end = nextSection?.index === undefined
|
||
? content.length
|
||
: afterHeading + nextSection.index;
|
||
return content.slice(start, end).trim();
|
||
}
|
||
|
||
function extractPreambleBeforeWorkflow(content: string, workflowMarkers: string[]): string {
|
||
const markerIndexes = workflowMarkers
|
||
.map(marker => content.indexOf(marker))
|
||
.filter(index => index >= 0);
|
||
expect(markerIndexes.length).toBeGreaterThan(0);
|
||
return content.slice(0, Math.min(...markerIndexes));
|
||
}
|
||
|
||
function isRepoRootSymlink(candidateDir: string): boolean {
|
||
try {
|
||
return fs.realpathSync(candidateDir) === fs.realpathSync(ROOT);
|
||
} catch {
|
||
return false;
|
||
}
|
||
}
|
||
|
||
// Dynamic template discovery — matches the generator's findTemplates() behavior.
|
||
// New skills automatically get test coverage without updating a static list.
|
||
const ALL_SKILLS = (() => {
|
||
const skills: Array<{ dir: string; name: string }> = [];
|
||
if (fs.existsSync(path.join(ROOT, 'SKILL.md.tmpl'))) {
|
||
skills.push({ dir: '.', name: 'root gstack' });
|
||
}
|
||
for (const entry of fs.readdirSync(ROOT, { withFileTypes: true })) {
|
||
if (!entry.isDirectory() || entry.name.startsWith('.') || entry.name === 'node_modules') continue;
|
||
if (fs.existsSync(path.join(ROOT, entry.name, 'SKILL.md.tmpl'))) {
|
||
skills.push({ dir: entry.name, name: entry.name });
|
||
}
|
||
}
|
||
return skills;
|
||
})();
|
||
|
||
// The claude host deliberately skips some skills (skipSkills — e.g. the
|
||
// /claude-code outside-voice skill exists only for non-Claude hosts), so those
|
||
// dirs have a SKILL.md.tmpl but no generated claude-host SKILL.md on a fresh
|
||
// checkout. Every generated-file assertion must exclude them or it is red on
|
||
// every clean clone (it was, invisibly, until the free suite ran in CI).
|
||
import { getHostConfig as __getHostConfig } from '../hosts/index';
|
||
const CLAUDE_SKIPPED = new Set(__getHostConfig('claude').generation.skipSkills ?? []);
|
||
const CLAUDE_GENERATED_SKILLS = ALL_SKILLS.filter(s => !CLAUDE_SKIPPED.has(s.dir));
|
||
|
||
// ─── Out-dir render isolation ────────────────────────────────
|
||
// Every generator invocation in this file that used to regenerate the live
|
||
// tree (the gitignored .agents/.factory/... host dirs included) now renders
|
||
// into this module-level out-dir: ONE `--host all` render covers the claude
|
||
// host plus every external host, and all golden-artifact reads plus the
|
||
// per-host `--dry-run` determinism checks point here. The tracked tree is
|
||
// only ever READ (the `generated files are fresh` dry-run deliberately
|
||
// compares against the committed files — that is a read, not a write).
|
||
// Out-dir renders of external hosts are byte-identical to in-place renders
|
||
// (pinned by test/gen-skill-docs-out-dir.test.ts).
|
||
const EXTERNAL_OUT = fs.mkdtempSync(path.join(os.tmpdir(), 'gstack-gen-docs-out-'));
|
||
{
|
||
const render = Bun.spawnSync(
|
||
['bun', 'run', 'scripts/gen-skill-docs.ts', '--host', 'all', '--out-dir', EXTERNAL_OUT],
|
||
{ cwd: ROOT, stdout: 'pipe', stderr: 'pipe', timeout: 120_000 },
|
||
);
|
||
if (render.exitCode !== 0) {
|
||
throw new Error(
|
||
`gen-skill-docs --host all --out-dir failed (exit ${render.exitCode}):\n${render.stderr.toString()}`,
|
||
);
|
||
}
|
||
}
|
||
afterAll(() => {
|
||
fs.rmSync(EXTERNAL_OUT, { recursive: true, force: true });
|
||
});
|
||
|
||
describe('gen-skill-docs', () => {
|
||
// Browse carve (token-reduction Phase 4): the command reference + snapshot
|
||
// flags render into browse/sections/command-list.md now — read the
|
||
// skeleton+sections union so these pins hold across the carve.
|
||
test('generated SKILL.md contains all command categories', () => {
|
||
const content = readSkillUnion('browse');
|
||
const categories = new Set(Object.values(COMMAND_DESCRIPTIONS).map(d => d.category));
|
||
for (const cat of categories) {
|
||
expect(content).toContain(`### ${cat}`);
|
||
}
|
||
});
|
||
|
||
test('generated SKILL.md contains all commands', () => {
|
||
const content = readSkillUnion('browse');
|
||
for (const [cmd, meta] of Object.entries(COMMAND_DESCRIPTIONS)) {
|
||
const display = meta.usage || cmd;
|
||
expect(content).toContain(display);
|
||
}
|
||
});
|
||
|
||
test('command table is sorted alphabetically within categories', () => {
|
||
const content = readSkillUnion('browse');
|
||
// Extract command names from the Navigation section as a test
|
||
const navSection = content.match(/### Navigation\n\|.*\n\|.*\n([\s\S]*?)(?=\n###|\n## )/);
|
||
expect(navSection).not.toBeNull();
|
||
const rows = navSection![1].trim().split('\n');
|
||
const commands = rows.map(r => {
|
||
const match = r.match(/\| `(\w+)/);
|
||
return match ? match[1] : '';
|
||
}).filter(Boolean);
|
||
const sorted = [...commands].sort();
|
||
expect(commands).toEqual(sorted);
|
||
});
|
||
|
||
test('snapshot flags section contains all flags', () => {
|
||
const content = readSkillUnion('browse');
|
||
for (const flag of SNAPSHOT_FLAGS) {
|
||
expect(content).toContain(flag.short);
|
||
expect(content).toContain(flag.description);
|
||
}
|
||
});
|
||
|
||
test('every skill has a SKILL.md.tmpl template', () => {
|
||
for (const skill of ALL_SKILLS) {
|
||
const tmplPath = path.join(ROOT, skill.dir, 'SKILL.md.tmpl');
|
||
expect(fs.existsSync(tmplPath)).toBe(true);
|
||
}
|
||
});
|
||
|
||
test('every skill has a generated SKILL.md with auto-generated header', () => {
|
||
for (const skill of CLAUDE_GENERATED_SKILLS) {
|
||
const mdPath = path.join(ROOT, skill.dir, 'SKILL.md');
|
||
expect(fs.existsSync(mdPath)).toBe(true);
|
||
const content = fs.readFileSync(mdPath, 'utf-8');
|
||
expect(content).toContain('AUTO-GENERATED from SKILL.md.tmpl');
|
||
expect(content).toContain('Regenerate: bun run gen:skill-docs');
|
||
}
|
||
});
|
||
|
||
// #1778: strict YAML parsers (Codex/OpenAI skill loading) reject frontmatter
|
||
// whose plain `description:` scalar contains an interior ": " (read as a nested
|
||
// mapping). Parse EVERY generated frontmatter block with a strict YAML parser,
|
||
// not just string-check that name:/description: exist.
|
||
test('every generated SKILL.md frontmatter parses as strict YAML', () => {
|
||
for (const skill of CLAUDE_GENERATED_SKILLS) {
|
||
const content = fs.readFileSync(path.join(ROOT, skill.dir, 'SKILL.md'), 'utf-8');
|
||
expect(validateSkillFrontmatter(content),
|
||
`frontmatter for ${skill.dir} must be valid YAML with string name/description`).toEqual([]);
|
||
}
|
||
});
|
||
|
||
test('every generated Codex (.agents/skills) frontmatter parses as strict YAML', () => {
|
||
// Reads the module-level out-dir render (guaranteed present — the render
|
||
// throws at module load if it fails), never the live gitignored tree.
|
||
const agentsDir = path.join(EXTERNAL_OUT, '.agents', 'skills');
|
||
for (const entry of fs.readdirSync(agentsDir, { withFileTypes: true })) {
|
||
if (!entry.isDirectory()) continue;
|
||
const mdPath = path.join(agentsDir, entry.name, 'SKILL.md');
|
||
if (!fs.existsSync(mdPath)) continue;
|
||
expect(validateSkillFrontmatter(fs.readFileSync(mdPath, 'utf-8')),
|
||
`Codex frontmatter for ${entry.name} must be valid YAML with string name/description`).toEqual([]);
|
||
}
|
||
});
|
||
|
||
test(`every generated SKILL.md description stays within ${MAX_SKILL_DESCRIPTION_LENGTH} chars`, () => {
|
||
for (const skill of CLAUDE_GENERATED_SKILLS) {
|
||
const content = fs.readFileSync(path.join(ROOT, skill.dir, 'SKILL.md'), 'utf-8');
|
||
const description = extractDescription(content);
|
||
expect(description.length).toBeLessThanOrEqual(MAX_SKILL_DESCRIPTION_LENGTH);
|
||
}
|
||
});
|
||
|
||
test(`every Codex SKILL.md description stays within ${MAX_SKILL_DESCRIPTION_LENGTH} chars`, () => {
|
||
const agentsDir = path.join(EXTERNAL_OUT, '.agents', 'skills');
|
||
for (const entry of fs.readdirSync(agentsDir, { withFileTypes: true })) {
|
||
if (!entry.isDirectory()) continue;
|
||
const skillMd = path.join(agentsDir, entry.name, 'SKILL.md');
|
||
if (!fs.existsSync(skillMd)) continue;
|
||
const content = fs.readFileSync(skillMd, 'utf-8');
|
||
const description = extractDescription(content);
|
||
expect(description.length).toBeLessThanOrEqual(MAX_SKILL_DESCRIPTION_LENGTH);
|
||
}
|
||
});
|
||
|
||
test('every Codex SKILL.md description stays under 900-char warning threshold', () => {
|
||
const WARN_THRESHOLD = 900;
|
||
const agentsDir = path.join(EXTERNAL_OUT, '.agents', 'skills');
|
||
const violations: string[] = [];
|
||
for (const entry of fs.readdirSync(agentsDir, { withFileTypes: true })) {
|
||
if (!entry.isDirectory()) continue;
|
||
const skillMd = path.join(agentsDir, entry.name, 'SKILL.md');
|
||
if (!fs.existsSync(skillMd)) continue;
|
||
const content = fs.readFileSync(skillMd, 'utf-8');
|
||
const description = extractDescription(content);
|
||
if (description.length > WARN_THRESHOLD) {
|
||
violations.push(`${entry.name}: ${description.length} chars (limit ${MAX_SKILL_DESCRIPTION_LENGTH}, ${MAX_SKILL_DESCRIPTION_LENGTH - description.length} remaining)`);
|
||
}
|
||
}
|
||
expect(violations).toEqual([]);
|
||
});
|
||
|
||
test('package.json version matches VERSION file (npm-valid translation)', () => {
|
||
// Decision 11 (v1.67 wave): VERSION stays the 4-digit source of truth;
|
||
// package.json carries the npm-valid 3-digit translation (npm rejects a
|
||
// fourth component). The pre-v1.67 1:1 four-digit mirror is also accepted
|
||
// (grandfathered until the next write), matching gstack-version-bump's
|
||
// own drift contract.
|
||
const pkg = JSON.parse(fs.readFileSync(path.join(ROOT, 'package.json'), 'utf-8'));
|
||
const version = fs.readFileSync(path.join(ROOT, 'VERSION'), 'utf-8').trim();
|
||
const npmTranslation = version.split('.').slice(0, 3).join('.');
|
||
expect([npmTranslation, version]).toContain(pkg.version);
|
||
});
|
||
|
||
test('generated files are fresh (match --dry-run)', () => {
|
||
// Deliberately compares against the LIVE TRACKED SKILL.md files (no
|
||
// --out-dir): this is the freshness gate for the committed tree. Dry-run
|
||
// writes nothing — it is a read.
|
||
const result = Bun.spawnSync(['bun', 'run', 'scripts/gen-skill-docs.ts', '--dry-run'], {
|
||
cwd: ROOT,
|
||
stdout: 'pipe',
|
||
stderr: 'pipe',
|
||
timeout: 120_000,
|
||
});
|
||
expect(result.exitCode).toBe(0);
|
||
const output = result.stdout.toString();
|
||
// Every skill should be FRESH
|
||
for (const skill of CLAUDE_GENERATED_SKILLS) {
|
||
const file = skill.dir === '.' ? 'SKILL.md' : `${skill.dir}/SKILL.md`;
|
||
expect(output).toContain(`FRESH: ${file}`);
|
||
}
|
||
expect(output).not.toContain('STALE');
|
||
});
|
||
|
||
test('no generated SKILL.md contains unresolved placeholders', () => {
|
||
for (const skill of CLAUDE_GENERATED_SKILLS) {
|
||
const content = fs.readFileSync(path.join(ROOT, skill.dir, 'SKILL.md'), 'utf-8');
|
||
const unresolved = content.match(/\{\{\w+\}\}/g);
|
||
expect(unresolved).toBeNull();
|
||
}
|
||
});
|
||
|
||
test('templates contain placeholders', () => {
|
||
// P2 (v1.2.0): the root template is a pure router — only {{PREAMBLE}}.
|
||
// The browse command/snapshot placeholders live in browse/SKILL.md.tmpl now.
|
||
const rootTmpl = fs.readFileSync(path.join(ROOT, 'SKILL.md.tmpl'), 'utf-8');
|
||
expect(rootTmpl).toContain('{{PREAMBLE}}');
|
||
expect(rootTmpl).not.toContain('{{COMMAND_REFERENCE}}');
|
||
expect(rootTmpl).not.toContain('{{SNAPSHOT_FLAGS}}');
|
||
|
||
// Browse carve: the reference resolvers moved into the on-demand section
|
||
// template (so gen-skill-docs keeps them fresh from browse/src); the
|
||
// skeleton points at the section instead of inlining the reference.
|
||
const browseTmpl = fs.readFileSync(path.join(ROOT, 'browse', 'SKILL.md.tmpl'), 'utf-8');
|
||
expect(browseTmpl).not.toContain('{{COMMAND_REFERENCE}}');
|
||
expect(browseTmpl).not.toContain('{{SNAPSHOT_FLAGS}}');
|
||
expect(browseTmpl).toContain('{{SECTION:command-list}}');
|
||
expect(browseTmpl).toContain('{{PREAMBLE}}');
|
||
|
||
const browseSectionTmpl = fs.readFileSync(
|
||
path.join(ROOT, 'browse', 'sections', 'command-list.md.tmpl'), 'utf-8');
|
||
expect(browseSectionTmpl).toContain('{{COMMAND_REFERENCE}}');
|
||
expect(browseSectionTmpl).toContain('{{SNAPSHOT_FLAGS}}');
|
||
|
||
// Aside is the primary browser: every browsing skill renders the Aside
|
||
// contract ({{ASIDE_SETUP}}); the browse binary is its fallback.
|
||
const qaTmpl = fs.readFileSync(path.join(ROOT, 'qa', 'SKILL.md.tmpl'), 'utf-8');
|
||
expect(qaTmpl).not.toContain('{{ASIDE_SETUP}}');
|
||
expect(qaTmpl).toContain('{{SECTION:browser-setup}}');
|
||
expect(fs.readFileSync(path.join(ROOT, 'qa/sections/browser-setup.md.tmpl'), 'utf8'))
|
||
.toContain('{{ASIDE_SETUP}}');
|
||
expect(browseTmpl).toContain('{{ASIDE_SETUP}}');
|
||
});
|
||
|
||
test('generated SKILL.md contains operational self-improvement (replaced contributor mode)', () => {
|
||
const content = fs.readFileSync(path.join(ROOT, 'SKILL.md'), 'utf-8');
|
||
expect(content).not.toContain('Contributor Mode');
|
||
expect(content).not.toContain('gstack_contributor');
|
||
expect(content).not.toContain('contributor-logs');
|
||
expect(content).toContain('Operational Self-Improvement');
|
||
expect(content).toContain('gstack-learnings-log');
|
||
// The learnings-resurface call moved from the inline preamble bash into
|
||
// the skill-start script (Phase 1) — same command, new home.
|
||
expect(SKILL_START_SCRIPT).toContain('gstack-learnings-search" --limit 3');
|
||
});
|
||
|
||
test('generated SKILL.md with LEARNINGS_LOG contains operational type', () => {
|
||
// Check a skill that has LEARNINGS_LOG (e.g., review)
|
||
const content = fs.readFileSync(path.join(ROOT, 'review', 'SKILL.md'), 'utf-8');
|
||
expect(content).toContain('operational');
|
||
});
|
||
|
||
test('session awareness lives in gstack-skill-start (registry touch + stale cleanup)', () => {
|
||
// The sessions registry moved from inline preamble bash into the script:
|
||
// it records the harness pid (--parent-pid identity) and expires entries
|
||
// older than 120 minutes.
|
||
expect(SKILL_START_SCRIPT).toContain('sessions/$PARENT_PID');
|
||
expect(SKILL_START_SCRIPT).toContain('-mmin +120');
|
||
// The render keeps the completion-status protocol the sessions feed into.
|
||
const content = fs.readFileSync(path.join(ROOT, 'SKILL.md'), 'utf-8');
|
||
expect(content).toContain('RECOMMENDATION');
|
||
});
|
||
|
||
test('branch detection lives in gstack-skill-start and is echoed as BRANCH', () => {
|
||
expect(SKILL_START_SCRIPT).toContain('_BRANCH=$(git branch --show-current');
|
||
expect(SKILL_START_SCRIPT).toContain('echo "BRANCH: $_BRANCH"');
|
||
});
|
||
|
||
// #2001: update_check: false silences the binary but the upgrade-handling
|
||
// instruction prose used to ship unconditionally. Token-reduction Phase 2
|
||
// made the gate STRUCTURAL: the prose left the renders entirely (absence is
|
||
// pinned by test/onboarding-moved-literals.test.ts) and now emits from
|
||
// gstack-skill-start's instruction layer ONLY when the update-check binary
|
||
// produced output — and that binary silences itself on update_check=false.
|
||
// Opted-out installs can never see the prose, by construction.
|
||
test('update_check opt-out gates the update binary and upgrade-flow emission (issue #2001)', () => {
|
||
// The config-echo cluster lives in gstack-skill-start: the flag is still
|
||
// read and echoed as a STATUS line for the model.
|
||
expect(SKILL_START_SCRIPT, 'script must read update_check config').toContain('_UPDATE_CHECK=$(');
|
||
expect(SKILL_START_SCRIPT, 'script must echo UPDATE_CHECK').toContain('echo "UPDATE_CHECK: $_UPDATE_CHECK"');
|
||
// Gate half 1: the update-check binary exits silently when opted out.
|
||
const updateCheck = fs.readFileSync(path.join(ROOT, 'bin', 'gstack-update-check'), 'utf-8');
|
||
expect(updateCheck, 'binary must read update_check config').toContain('get update_check');
|
||
expect(updateCheck, 'binary must exit silently on update_check=false')
|
||
.toMatch(/if \[ "\$_UC" = "false" \]; then\n\s*exit 0/);
|
||
// Gate half 2: the upgrade-flow instruction block emits only when the
|
||
// binary emitted something (empty when opted out, cached, or up to date).
|
||
expect(SKILL_START_SCRIPT, 'upgrade-flow must be gated on update-check output')
|
||
.toMatch(/if \[ -n "\$_UPD" \]; then\n\s*_emit_block upgrade-flow/);
|
||
});
|
||
|
||
test('tier 2+ skills contain ELI10 simplification rules (AskUserQuestion format)', () => {
|
||
// Root SKILL.md is tier 1 and CSO intentionally uses a private startup with
|
||
// no shared PREAMBLE. Check a regular tier 2+ PREAMBLE consumer instead.
|
||
// v1.7.0.0 Pros/Cons format uses "ELI10 (ALWAYS)" rather than "Simplify (ELI10".
|
||
const content = fs.readFileSync(path.join(ROOT, 'review', 'SKILL.md'), 'utf-8');
|
||
expect(content).toContain('ELI10');
|
||
expect(content).toContain('plain English');
|
||
expect(content).toContain('not function names');
|
||
});
|
||
|
||
test('tier 1 skills do NOT contain AskUserQuestion format', () => {
|
||
// Use benchmark (tier 1) instead of root — root SKILL.md gets overwritten by Codex test setup
|
||
const content = fs.readFileSync(path.join(ROOT, 'benchmark', 'SKILL.md'), 'utf-8');
|
||
expect(content).not.toContain('## AskUserQuestion Format');
|
||
expect(content).not.toContain('## Completeness Principle');
|
||
});
|
||
|
||
test('telemetry producer lives in the scripts; render documents the analytics sink', () => {
|
||
// The skill-usage.jsonl producers moved into the scripts (Phase 1).
|
||
expect(SKILL_START_SCRIPT).toContain('analytics/skill-usage.jsonl');
|
||
expect(SKILL_END_SCRIPT).toContain('analytics/skill-usage.jsonl');
|
||
// The render still tells the model where telemetry lands.
|
||
const content = fs.readFileSync(path.join(ROOT, 'SKILL.md'), 'utf-8');
|
||
expect(content).toContain('~/.gstack/analytics');
|
||
});
|
||
|
||
test('plan-review generated preambles stay under the Option A budget', () => {
|
||
const reviewSkills = [
|
||
{
|
||
path: path.join(ROOT, 'plan-ceo-review', 'SKILL.md'),
|
||
markers: ['# Mega Plan Review Mode', '## Step 0: Detect platform and base branch'],
|
||
},
|
||
{
|
||
path: path.join(ROOT, 'plan-eng-review', 'SKILL.md'),
|
||
markers: ['# Plan Review Mode'],
|
||
},
|
||
];
|
||
|
||
// Plan skills carry the same preamble surface as other tier-≥2 skills
|
||
// (Artifacts Sync, Context Recovery, Routing Injection are load-bearing
|
||
// functionality, not optional). Budget is set to current size + small
|
||
// headroom; ratchet down if a future slim trims real bytes.
|
||
// Ratcheted from 33000 → 35000 when the gbrain context-load block was
|
||
// added (per /sync-gbrain plan §4). Ratcheted 35000 → 36500 in v1.27.0.0
|
||
// when generate-brain-sync-block.ts gained the gbrain_mcp_mode probe +
|
||
// remote-mode ARTIFACTS_SYNC status line (Path 4 of /setup-gbrain).
|
||
// Ratcheted 36500 → 39000 in the contributor wave when #1205 added the
|
||
// \\u-escape CJK rule (rule 12 + self-check item) to the AskUserQuestion
|
||
// preamble.
|
||
// Ratcheted 39000 → 40000 in plan-tune cathedral T14: question-tuning
|
||
// resolver gained the <gstack-qid:...> marker convention + the
|
||
// (recommended) label requirement (D2 + D18 — both load-bearing for
|
||
// hook enforcement). Adds ~700 bytes.
|
||
// Ratcheted 40000 → 60000 in v1.52.0.0 cap audit: ~20K headroom so
|
||
// future preamble adds don't trip the gate on each PR. Real runaway
|
||
// (preamble doubling) still trips; normal scope growth doesn't.
|
||
for (const skill of reviewSkills) {
|
||
const content = fs.readFileSync(skill.path, 'utf-8');
|
||
const preamble = extractPreambleBeforeWorkflow(content, skill.markers);
|
||
expect(Buffer.byteLength(preamble, 'utf-8')).toBeLessThan(60_000);
|
||
}
|
||
});
|
||
|
||
test('voice and writing-style preamble sections stay compact', () => {
|
||
const content = readSkillUnion('plan-eng-review'); // carved: review body moved to section
|
||
const voice = extractMarkdownSection(content, '## Voice');
|
||
const writingStyle = extractMarkdownSection(content, '## Writing Style');
|
||
|
||
expect(Buffer.byteLength(voice, 'utf-8')).toBeLessThan(3_000);
|
||
expect(Buffer.byteLength(writingStyle, 'utf-8')).toBeLessThan(2_000);
|
||
});
|
||
|
||
test('slim voice section preserves the gstack voice contract', () => {
|
||
const content = readSkillUnion('plan-eng-review'); // carved: review body moved to section
|
||
const voice = extractMarkdownSection(content, '## Voice');
|
||
|
||
expect(voice).toMatch(/lead with the point|direct/i);
|
||
expect(voice).toMatch(/file|function|line|command|real numbers/i);
|
||
expect(voice).toMatch(/user.*outcome|user.*experience|real user/i);
|
||
expect(voice).toMatch(/corporate|academic|PR|hype/i);
|
||
expect(voice).toMatch(/AI vocabulary|delve|crucial|robust/i);
|
||
expect(voice).toMatch(/user decides|user.*context|sovereignty|recommendation, not a decision/i);
|
||
});
|
||
|
||
test('preamble .pending-* glob is zsh-safe (uses find, not shell glob)', () => {
|
||
for (const skill of CLAUDE_GENERATED_SKILLS) {
|
||
const content = fs.readFileSync(path.join(ROOT, skill.dir, 'SKILL.md'), 'utf-8');
|
||
if (!content.includes('.pending-')) continue;
|
||
// Must NOT have a bare shell glob ".pending-*" outside of find's -name argument
|
||
expect(content).not.toMatch(/for _PF in [^\n]*\/\.pending-\*/);
|
||
// Must use find to avoid zsh NOMATCH error on glob expansion
|
||
expect(content).toContain("find ~/.gstack/analytics -maxdepth 1 -name '.pending-*'");
|
||
}
|
||
});
|
||
|
||
test('bash blocks with shell globs are zsh-safe (setopt guard or find)', () => {
|
||
for (const skill of CLAUDE_GENERATED_SKILLS) {
|
||
const content = fs.readFileSync(path.join(ROOT, skill.dir, 'SKILL.md'), 'utf-8');
|
||
const bashBlocks = [...content.matchAll(/```bash\n([\s\S]*?)```/g)].map(m => m[1]);
|
||
|
||
for (const block of bashBlocks) {
|
||
const lines = block.split('\n');
|
||
|
||
for (const line of lines) {
|
||
const trimmed = line.trimStart();
|
||
if (trimmed.startsWith('#')) continue;
|
||
if (!trimmed.includes('*')) continue;
|
||
// Skip lines where * is inside find -name, git pathspecs, or $(find)
|
||
if (/\bfind\b/.test(trimmed)) continue;
|
||
if (/\bgit\b/.test(trimmed)) continue;
|
||
if (/\$\(find\b/.test(trimmed)) continue;
|
||
|
||
// Check 1: "for VAR in <glob>" must use $(find ...) — caught above by the
|
||
// $(find check, so any surviving for-in with a glob pattern is a violation
|
||
if (/\bfor\s+\w+\s+in\b/.test(trimmed) && /\*\./.test(trimmed)) {
|
||
throw new Error(
|
||
`Unsafe for-in glob in ${skill.dir}/SKILL.md: "${trimmed}". ` +
|
||
`Use \`for f in $(find ... -name '*.ext')\` for zsh compatibility.`
|
||
);
|
||
}
|
||
|
||
// Check 2: ls/cat/rm/grep with glob file args must have setopt guard
|
||
const isGlobCmd = /\b(?:ls|cat|rm|grep)\b/.test(trimmed) &&
|
||
/(?:\/\*[a-z.*]|\*\.[a-z])/.test(trimmed);
|
||
if (isGlobCmd) {
|
||
expect(block).toContain('setopt +o nomatch');
|
||
}
|
||
}
|
||
}
|
||
}
|
||
});
|
||
|
||
test('preamble-using skills have correct skill name in telemetry', () => {
|
||
const PREAMBLE_SKILLS = [
|
||
{ dir: '.', name: 'gstack' },
|
||
{ dir: 'ship', name: 'ship' },
|
||
{ dir: 'review', name: 'review' },
|
||
{ dir: 'qa', name: 'qa' },
|
||
{ dir: 'retro', name: 'retro' },
|
||
];
|
||
for (const skill of PREAMBLE_SKILLS) {
|
||
const content = fs.readFileSync(path.join(ROOT, skill.dir, 'SKILL.md'), 'utf-8');
|
||
// The skill name now travels as --skill into gstack-skill-start (the
|
||
// preamble fence) and gstack-skill-end (the telemetry epilogue) — the
|
||
// scripts write it into the JSONL events.
|
||
expect(content, `${skill.dir} preamble fence must pass its own name`)
|
||
.toMatch(new RegExp(`--skill "${skill.name}" --model`));
|
||
expect(content, `${skill.dir} epilogue must pass its own name`)
|
||
.toContain(`gstack-skill-end --skill "${skill.name}"`);
|
||
}
|
||
});
|
||
|
||
test('qa and qa-only load the shared QA_METHODOLOGY through exploration', () => {
|
||
const qaSkeletonTmpl = fs.readFileSync(path.join(ROOT, 'qa', 'SKILL.md.tmpl'), 'utf-8');
|
||
expect(qaSkeletonTmpl).toContain('{{SECTION:exploratory}}');
|
||
expect(qaSkeletonTmpl).not.toContain('{{QA_METHOD_READS}}');
|
||
expect(qaSkeletonTmpl).toContain("Follow the shared section's ordered preparation");
|
||
expect(qaSkeletonTmpl).not.toContain('{{QA_METHODOLOGY}}');
|
||
const qaSectionTmpl = fs.readFileSync(path.join(ROOT, 'qa', 'sections', 'qa-patterns.md.tmpl'), 'utf-8');
|
||
expect(qaSectionTmpl).toContain('{{QA_METHODOLOGY}}');
|
||
|
||
const qaOnlyTmpl = fs.readFileSync(path.join(ROOT, 'qa-only', 'SKILL.md.tmpl'), 'utf-8');
|
||
expect(qaOnlyTmpl).not.toContain('{{QA_METHODOLOGY}}');
|
||
expect(qaOnlyTmpl).toContain('{{SECTION:exploratory}}');
|
||
expect(qaOnlyTmpl).not.toContain('{{QA_METHOD_READS}}');
|
||
expect(qaOnlyTmpl).toContain('Load the shared preparation gate now');
|
||
expect(qaOnlyTmpl).toContain('Use the shared section already loaded above');
|
||
for (const skill of ['qa', 'qa-only']) {
|
||
expect(fs.readFileSync(path.join(ROOT, skill, 'sections/exploratory.md.tmpl'), 'utf8'))
|
||
.toContain('{{QA_EXPLORATORY}}');
|
||
const entry = fs.readFileSync(path.join(ROOT, skill, 'SKILL.md'), 'utf8');
|
||
const explorer = fs.readFileSync(path.join(ROOT, skill, 'sections/exploratory.md'), 'utf8');
|
||
const directoryRead = skill === 'qa'
|
||
? 'Read `sections/scope.md` relative to the installed `qa`/`gstack-qa` SKILL.md directory'
|
||
: "Use this host's installed `qa`/`gstack-qa` SKILL.md directory for these reads";
|
||
expect(entry).toContain('sections/exploratory.md');
|
||
expect(explorer).toContain(directoryRead);
|
||
for (const method of ['system-functional', 'qa-patterns']) {
|
||
const methodRead = `Read \`sections/${method}.md\` in full.`;
|
||
expect(entry).not.toContain(methodRead);
|
||
expect(explorer).toContain(methodRead);
|
||
expect(explorer.split(methodRead)).toHaveLength(2);
|
||
expect(explorer.indexOf(directoryRead)).toBeLessThan(explorer.indexOf(methodRead));
|
||
const directory = path.resolve(ROOT, skill, skill === 'qa-only' ? '../qa' : '.');
|
||
expect(fs.realpathSync(path.join(directory, 'sections', `${method}.md`)))
|
||
.toBe(path.join(ROOT, 'qa', 'sections', `${method}.md`));
|
||
}
|
||
expect(explorer).toContain('**Browser surfaces only:**\nRead `sections/qa-patterns.md` in full.');
|
||
expect(explorer).toContain('Complete these Reads in order before writing charters or probing');
|
||
expect(explorer).toContain('Do not repeat a Read already completed in this invocation');
|
||
expect(explorer.indexOf('Read `sections/qa-patterns.md`')).toBeLessThan(explorer.indexOf('Write a **charter**'));
|
||
}
|
||
});
|
||
|
||
test('QA_METHODOLOGY is expanded in the shared resource referenced by qa and qa-only', () => {
|
||
const qaContent = readSkillUnion('qa'); // carved: methodology lives in qa/sections/qa-patterns.md
|
||
const qaOnlyUnion = readSkillUnion('qa-only');
|
||
expect(qaOnlyUnion).toContain("Use this host's installed `qa`/`gstack-qa` SKILL.md directory for these reads");
|
||
expect(qaOnlyUnion).toContain('**Browser surfaces only:**\nRead `sections/qa-patterns.md` in full.');
|
||
expect(qaOnlyUnion).not.toContain('Health Score Rubric');
|
||
const qaOnlyContent = qaOnlyUnion + fs.readFileSync(path.join(ROOT, 'qa/sections/qa-patterns.md'), 'utf8');
|
||
|
||
// Both should contain the health score rubric
|
||
expect(qaContent).toContain('Health Score Rubric');
|
||
expect(qaOnlyContent).toContain('Health Score Rubric');
|
||
|
||
// Both should contain framework guidance
|
||
expect(qaContent).toContain('Framework-Specific Guidance');
|
||
expect(qaOnlyContent).toContain('Framework-Specific Guidance');
|
||
|
||
// Both should contain the important rules
|
||
expect(qaContent).toContain('Important Rules');
|
||
expect(qaOnlyContent).toContain('Important Rules');
|
||
|
||
// Both should contain the 6 phases
|
||
expect(qaContent).toContain('Phase 1');
|
||
expect(qaOnlyContent).toContain('Phase 1');
|
||
expect(qaContent).toContain('Phase 6');
|
||
expect(qaOnlyContent).toContain('Phase 6');
|
||
});
|
||
|
||
test('qa-only has no-fix guardrails', () => {
|
||
const qaOnlyContent = fs.readFileSync(path.join(ROOT, 'qa-only', 'SKILL.md'), 'utf-8');
|
||
expect(qaOnlyContent).toContain('Never fix bugs');
|
||
expect(qaOnlyContent).toContain('NEVER fix anything');
|
||
// Should not have Edit, Glob, or Grep in allowed-tools.
|
||
// Scope to frontmatter (between the first two --- lines) — the body can
|
||
// legitimately mention these tool names in prose (e.g., Claude model
|
||
// overlay says "prefer Read, Edit, Write, Glob, Grep over Bash").
|
||
const fmMatch = qaOnlyContent.match(/^---\n([\s\S]*?)\n---/);
|
||
expect(fmMatch).not.toBeNull();
|
||
const frontmatter = fmMatch![1];
|
||
expect(frontmatter).toMatch(/allowed-tools:/);
|
||
expect(frontmatter).not.toMatch(/allowed-tools:[\s\S]*?- Edit/);
|
||
expect(frontmatter).not.toMatch(/allowed-tools:[\s\S]*?- Glob/);
|
||
expect(frontmatter).not.toMatch(/allowed-tools:[\s\S]*?- Grep/);
|
||
});
|
||
|
||
test('qa has fix-loop tools and phases', () => {
|
||
const qaContent = fs.readFileSync(path.join(ROOT, 'qa', 'SKILL.md'), 'utf-8');
|
||
// Should have Edit, Glob, Grep in allowed-tools
|
||
expect(qaContent).toContain('Edit');
|
||
expect(qaContent).toContain('Glob');
|
||
expect(qaContent).toContain('Grep');
|
||
// Should have fix-loop phases
|
||
expect(qaContent).toContain('Phase 7');
|
||
expect(qaContent).toContain('Phase 8');
|
||
expect(qaContent).toContain('Fix Loop');
|
||
expect(qaContent).toContain('Triage');
|
||
expect(qaContent).toContain('WTF');
|
||
});
|
||
});
|
||
|
||
describe('BASE_BRANCH_DETECT resolver', () => {
|
||
// Find a generated SKILL.md that uses the placeholder (ship is guaranteed to)
|
||
const shipContent = readShipUnion();
|
||
|
||
test('resolver output contains PR base detection command', () => {
|
||
expect(shipContent).toContain('gh pr view --json baseRefName');
|
||
});
|
||
|
||
test('resolver output contains repo default branch detection command', () => {
|
||
expect(shipContent).toContain('gh repo view --json defaultBranchRef');
|
||
});
|
||
|
||
test('resolver output contains fallback to main', () => {
|
||
expect(shipContent).toMatch(/fall\s*back\s+to\s+`main`/i);
|
||
});
|
||
|
||
test('resolver output uses "the base branch" phrasing', () => {
|
||
expect(shipContent).toContain('the base branch');
|
||
});
|
||
|
||
test('resolver output contains GitLab CLI commands', () => {
|
||
expect(shipContent).toContain('glab');
|
||
});
|
||
|
||
test('resolver output contains git-native fallback', () => {
|
||
expect(shipContent).toContain('git symbolic-ref');
|
||
});
|
||
|
||
test('resolver output mentions GitLab platform', () => {
|
||
expect(shipContent).toMatch(/gitlab/i);
|
||
});
|
||
});
|
||
|
||
describe('GitLab support in generated skills', () => {
|
||
const retroContent = fs.readFileSync(path.join(ROOT, 'retro', 'SKILL.md'), 'utf-8');
|
||
const shipSkillContent = readShipUnion();
|
||
|
||
test('retro contains GitLab MR number extraction', () => {
|
||
expect(retroContent).toContain('[#!]');
|
||
});
|
||
|
||
test('retro uses BASE_BRANCH_DETECT (contains glab)', () => {
|
||
expect(retroContent).toContain('glab');
|
||
});
|
||
|
||
test('ship contains glab mr create', () => {
|
||
expect(shipSkillContent).toContain('glab mr create');
|
||
});
|
||
|
||
test('ship checks .gitlab-ci.yml', () => {
|
||
expect(shipSkillContent).toContain('.gitlab-ci.yml');
|
||
});
|
||
});
|
||
|
||
/**
|
||
* Quality evals — catch description regressions.
|
||
*
|
||
* These test that generated output is *useful for an AI agent*,
|
||
* not just structurally valid. Each test targets a specific
|
||
* regression we actually shipped and caught in review.
|
||
*/
|
||
describe('description quality evals', () => {
|
||
// Regression: snapshot flags lost value hints (-d <N>, -s <sel>, -o <path>)
|
||
// Browse carve: the flag reference renders into browse/sections/command-list.md.
|
||
test('snapshot flags with values include value hints in output', () => {
|
||
const content = readSkillUnion('browse');
|
||
for (const flag of SNAPSHOT_FLAGS) {
|
||
if (flag.takesValue) {
|
||
expect(flag.valueHint).toBeDefined();
|
||
expect(content).toContain(`${flag.short} ${flag.valueHint}`);
|
||
}
|
||
}
|
||
});
|
||
|
||
// Regression: "is" lost the valid states enum
|
||
test('is command lists valid state values', () => {
|
||
const desc = COMMAND_DESCRIPTIONS['is'].description;
|
||
for (const state of ['visible', 'hidden', 'enabled', 'disabled', 'checked', 'editable', 'focused']) {
|
||
expect(desc).toContain(state);
|
||
}
|
||
});
|
||
|
||
// Regression: "press" lost common key examples
|
||
test('press command lists example keys', () => {
|
||
const desc = COMMAND_DESCRIPTIONS['press'].description;
|
||
expect(desc).toContain('Enter');
|
||
expect(desc).toContain('Tab');
|
||
expect(desc).toContain('Escape');
|
||
});
|
||
|
||
// Regression: "console" lost --errors filter note
|
||
test('console command describes --errors behavior', () => {
|
||
const desc = COMMAND_DESCRIPTIONS['console'].description;
|
||
expect(desc).toContain('--errors');
|
||
});
|
||
|
||
// Regression: snapshot -i lost "@e refs" context
|
||
test('snapshot -i mentions @e refs', () => {
|
||
const flag = SNAPSHOT_FLAGS.find(f => f.short === '-i')!;
|
||
expect(flag.description).toContain('@e');
|
||
});
|
||
|
||
// Regression: snapshot -C lost "@c refs" context
|
||
test('snapshot -C mentions @c refs', () => {
|
||
const flag = SNAPSHOT_FLAGS.find(f => f.short === '-C')!;
|
||
expect(flag.description).toContain('@c');
|
||
});
|
||
|
||
// Guard: every description must be at least 8 chars (catches empty or stub descriptions)
|
||
test('all command descriptions have meaningful length', () => {
|
||
for (const [cmd, meta] of Object.entries(COMMAND_DESCRIPTIONS)) {
|
||
expect(meta.description.length).toBeGreaterThanOrEqual(8);
|
||
}
|
||
});
|
||
|
||
// Guard: snapshot flag descriptions must be at least 10 chars
|
||
test('all snapshot flag descriptions have meaningful length', () => {
|
||
for (const flag of SNAPSHOT_FLAGS) {
|
||
expect(flag.description.length).toBeGreaterThanOrEqual(10);
|
||
}
|
||
});
|
||
|
||
// Guard: descriptions must not contain pipe (breaks markdown table cells)
|
||
// Usage strings are backtick-wrapped in the table so pipes there are safe.
|
||
test('no command description contains pipe character', () => {
|
||
for (const [cmd, meta] of Object.entries(COMMAND_DESCRIPTIONS)) {
|
||
expect(meta.description).not.toContain('|');
|
||
}
|
||
});
|
||
|
||
// Guard: generated output uses → not ->
|
||
test('generated SKILL.md uses unicode arrows', () => {
|
||
// P2 (v1.2.0): the browse body moved out of the top-level router into
|
||
// browse/SKILL.md. Guard arrow style on the browse body (sliced from its
|
||
// H1 so the auto-generated `-->` header comments are excluded).
|
||
const content = fs.readFileSync(path.join(ROOT, 'browse', 'SKILL.md'), 'utf-8');
|
||
const h1 = content.search(/^# browse: /m);
|
||
expect(h1).toBeGreaterThan(-1);
|
||
const body = content.slice(h1);
|
||
expect(body).toContain('→');
|
||
expect(body).not.toContain('->');
|
||
});
|
||
});
|
||
|
||
describe('REVIEW_DASHBOARD resolver', () => {
|
||
const REVIEW_SKILLS = ['plan-ceo-review', 'plan-eng-review', 'plan-design-review'];
|
||
|
||
for (const skill of REVIEW_SKILLS) {
|
||
test(`review dashboard appears in ${skill} generated file`, () => {
|
||
const content = readSkillUnion(skill); // carved skills: union skeleton + sections
|
||
expect(content).toContain('gstack-review');
|
||
expect(content).toContain('REVIEW READINESS DASHBOARD');
|
||
});
|
||
}
|
||
|
||
test('review dashboard appears in ship generated file', () => {
|
||
const content = readShipUnion();
|
||
expect(content).toContain('reviews.jsonl');
|
||
expect(content).toContain('REVIEW READINESS DASHBOARD');
|
||
});
|
||
|
||
test('dashboard treats review as a valid Eng Review source', () => {
|
||
const content = readShipUnion();
|
||
expect(content).toContain('| Eng Review | `review` or `plan-eng-review` | (DIFF) or (PLAN) |');
|
||
expect(content).toContain('**Content-first rule:** For `review`');
|
||
expect(content).toContain('**Plan records** (plan-ceo-review, plan-eng-review');
|
||
expect(content.replace(/\s+/g, ' ')).toContain('CLEARED requires the selected Eng Review to be `clean`, within 7 days and fresh under step 2');
|
||
});
|
||
|
||
test('shared dashboard propagates review source to plan-eng-review', () => {
|
||
const content = readSkillUnion('plan-eng-review'); // carved: review body moved to section
|
||
expect(content).toContain('| Eng Review | `review` or `plan-eng-review` | (DIFF) or (PLAN) |');
|
||
expect(content).toContain('**Content-first rule:** For `review`');
|
||
});
|
||
|
||
test('resolver output contains key dashboard elements', () => {
|
||
const content = readSkillUnion('plan-ceo-review'); // carved: dashboard moved to section
|
||
expect(content).toContain('VERDICT');
|
||
expect(content).toContain('CLEARED');
|
||
expect(content).toContain('Eng Review');
|
||
expect(content).toContain('7 days');
|
||
expect(content).toContain('Design Review');
|
||
expect(content).toContain('skip_eng_review');
|
||
});
|
||
|
||
test('dashboard bash block includes git HEAD for staleness detection', () => {
|
||
const content = readSkillUnion('plan-ceo-review'); // carved: dashboard moved to section
|
||
expect(content).toContain('git rev-parse --short HEAD');
|
||
expect(content).toContain('---HEAD---');
|
||
});
|
||
|
||
test('dashboard includes staleness detection prose', () => {
|
||
const content = readSkillUnion('plan-ceo-review'); // carved: dashboard moved to section
|
||
expect(content).toContain('**2. Check freshness before choosing a verdict.**');
|
||
expect(content).toContain('commit');
|
||
});
|
||
|
||
for (const skill of REVIEW_SKILLS) {
|
||
test(`${skill} contains review chaining section`, () => {
|
||
const content = readSkillUnion(skill); // carved skills: union skeleton + sections
|
||
expect(content).toContain('Review Chaining');
|
||
});
|
||
|
||
test(`${skill} Review Log includes commit field`, () => {
|
||
const content = readSkillUnion(skill); // carved skills: union skeleton + sections
|
||
expect(content).toContain('"commit"');
|
||
});
|
||
}
|
||
|
||
test('plan-ceo-review chaining mentions eng and design reviews', () => {
|
||
// Carved skill: the chaining prose lives in sections/*.md. (It used to
|
||
// pass against the skeleton only because the preamble's routing-injection
|
||
// rules incidentally named these skills — that prose moved into
|
||
// bin/gstack-skill-start in token-reduction Phase 2.)
|
||
const content = readSkillUnion('plan-ceo-review');
|
||
expect(content).toContain('/plan-eng-review');
|
||
expect(content).toContain('/plan-design-review');
|
||
});
|
||
|
||
test('plan-eng-review chaining mentions design and ceo reviews', () => {
|
||
const content = readSkillUnion('plan-eng-review'); // carved: review body moved to section
|
||
expect(content).toContain('/plan-design-review');
|
||
expect(content).toContain('/plan-ceo-review');
|
||
});
|
||
|
||
test('plan-design-review chaining mentions eng, ceo, and design skills', () => {
|
||
const content = readSkillUnion('plan-design-review');
|
||
expect(content).toContain('/plan-eng-review');
|
||
expect(content).toContain('/plan-ceo-review');
|
||
expect(content).toContain('/design-shotgun');
|
||
expect(content).toContain('/design-html');
|
||
});
|
||
|
||
test('ship does NOT contain review chaining', () => {
|
||
const content = readShipUnion();
|
||
expect(content).not.toContain('Review Chaining');
|
||
});
|
||
});
|
||
|
||
// ─── Test Coverage Audit Resolver Tests ─────────────────────
|
||
|
||
describe('TEST_COVERAGE_AUDIT placeholders', () => {
|
||
const planSkill = readSkillUnion('plan-eng-review'); // carved
|
||
const shipSkill = readShipUnion();
|
||
const reviewSkill = readSkillUnion('review'); // carved: Review Army moved to sections/review-army.md
|
||
|
||
test('plan and ship modes share codepath tracing methodology', () => {
|
||
// Review mode delegates test coverage to the Testing specialist subagent (Review Army)
|
||
const normalizedPlanSkill = planSkill.replace(/\s+/g, ' ');
|
||
const normalizedShipSkill = shipSkill.replace(/\s+/g, ' ');
|
||
const sharedPhrases = [
|
||
'Trace data flow',
|
||
'Diagram the execution',
|
||
'concrete source and test files',
|
||
'dedicated tool call before drawing the diagram',
|
||
'use separate calls for',
|
||
'context. Base the diagram on that read',
|
||
'start Test review output with the coverage diagram',
|
||
'In full plan reviews, put it inside the normal Test review section',
|
||
'Required outputs keep the final terminal report order',
|
||
'Avoid bare `[ ]` or `[x]` in diagrams',
|
||
'keep user-flow markers off code-path rows',
|
||
'Quality scoring rubric',
|
||
'★★★',
|
||
'★★',
|
||
'GAP',
|
||
];
|
||
for (const phrase of sharedPhrases) {
|
||
expect(normalizedPlanSkill).toContain(phrase);
|
||
expect(normalizedShipSkill).toContain(phrase);
|
||
}
|
||
expect(normalizedPlanSkill).toContain('For every target, run these five Test steps inside Section 3, after Scope Challenge and the Architecture/Code Quality reviews');
|
||
expect(normalizedPlanSkill).toContain('Within Test step 1, read concrete source/tests before tracing or diagramming; Test step 2 adds user flows');
|
||
expect(normalizedShipSkill).toContain('Finish this source read before tracing data flow in audit item 2 below; map user flows afterward');
|
||
// Plan mode traces the plan, not a git diff
|
||
expect(planSkill).toContain('Trace every codepath in the plan');
|
||
expect(planSkill).not.toContain('git diff origin');
|
||
// Ship mode traces the diff
|
||
expect(shipSkill).toContain('Trace every codepath changed');
|
||
});
|
||
|
||
test('review mode uses Review Army for specialist dispatch', () => {
|
||
expect(reviewSkill).toContain('Review Army');
|
||
expect(reviewSkill).toContain('Specialist Dispatch');
|
||
expect(reviewSkill).toContain('testing.md');
|
||
});
|
||
|
||
test('plan and ship modes include E2E decision matrix', () => {
|
||
// Review mode delegates to Testing specialist
|
||
for (const skill of [planSkill, shipSkill]) {
|
||
expect(skill).toContain('E2E Test Decision Matrix');
|
||
expect(skill).toContain('→E2E');
|
||
expect(skill).toContain('→EVAL');
|
||
}
|
||
});
|
||
|
||
test('plan and ship modes include regression rule', () => {
|
||
// Review mode delegates to Testing specialist
|
||
for (const skill of [planSkill, shipSkill]) {
|
||
expect(skill).toContain('REGRESSION RULE');
|
||
expect(skill).toContain('IRON RULE');
|
||
}
|
||
});
|
||
|
||
test('all five plan audit subheadings stay inside Test review while ship keeps its existing depth', () => {
|
||
const tests = planSkill.split('\n### 3. Test review\n')[1]!.split('\n### 4. Performance review\n')[0]!;
|
||
const instructions = tests.replace(/```[\s\S]*?```/g, '');
|
||
const headings = ['Test Framework Detection', 'E2E Test Decision Matrix', 'REGRESSION RULE (mandatory)',
|
||
'LLM/eval scope', 'Test Plan Artifact'];
|
||
expect([...instructions.matchAll(/^#### (.+)$/gm)].map(match => match[1])).toEqual(headings);
|
||
expect(instructions).not.toMatch(/^#{1,3} /m);
|
||
for (const heading of headings.filter(heading => heading !== 'LLM/eval scope')) {
|
||
expect(shipSkill).toContain(`\n### ${heading}\n`);
|
||
expect(shipSkill).not.toContain(`\n#### ${heading}\n`);
|
||
}
|
||
});
|
||
|
||
test('planned regression coverage gets its own approved contract without making coverage optional', () => {
|
||
const regression = planSkill.split('\n#### REGRESSION RULE (mandatory)\n')[1]!.split('**Step 4.')[0]!;
|
||
expect(regression).toContain('critical requirement');
|
||
expect(regression).toContain('Carry forward an exact approved regression contract; otherwise use one dedicated AskUserQuestion');
|
||
expect(regression).toContain('behavior to preserve, intentional changes, and acceptance assertions');
|
||
expect(regression).toContain('Ask how to cover it, not whether to skip it');
|
||
expect(regression).toContain('before adding the approved contract to the plan');
|
||
expect(regression).not.toContain('No AskUserQuestion');
|
||
expect(regression).not.toContain('the diff broke');
|
||
expect(regression).toContain('not proof that running code already broke');
|
||
});
|
||
|
||
test('plan gap additions wait for their test-contract decision', () => {
|
||
const action = planSkill.split('**Step 5. Add missing tests to the plan:**')[1]!.split('\n#### Test Plan Artifact\n')[0]!;
|
||
expect(action).toContain('Collect the requirements for each GAP and the LLM/eval scope above');
|
||
expect(action).toContain('Carry forward required proof of approved behavior');
|
||
expect(action).toContain('Mark new contracts and optional depth choices pending until the decision gate below resolves them');
|
||
expect(action).toContain('**STOP for each pending decision.**');
|
||
expect(action).toContain('Wait for its answer before applying that remedy');
|
||
expect(action).not.toContain('report the findings and their dispositions and continue');
|
||
expect(action).not.toContain('For each GAP identified in the diagram, add a test requirement');
|
||
expect(action).not.toContain('explain what broke');
|
||
});
|
||
|
||
test('implemented regressions still require an immediate test without a question', () => {
|
||
const regression = shipSkill.split('### REGRESSION RULE (mandatory)')[1]!.split('**4.')[0]!;
|
||
expect(regression).toContain('code that previously worked but the diff broke');
|
||
expect(regression).toContain('written immediately. No AskUserQuestion. No skipping.');
|
||
expect(regression).toContain('The change introduces a new failure mode for existing callers');
|
||
});
|
||
|
||
test('plan and ship modes include test framework detection', () => {
|
||
// Review mode delegates to Testing specialist
|
||
for (const skill of [planSkill, shipSkill]) {
|
||
expect(skill).toContain('Test Framework Detection');
|
||
expect(skill).toContain('CLAUDE.md');
|
||
}
|
||
});
|
||
|
||
test('plan framework absence keeps requirements without offering test generation', () => {
|
||
const detection = planSkill.split('\n#### Test Framework Detection\n')[1]!.split('**Step 1.')[0]!;
|
||
expect(detection).toContain('continue the diagram and planned assertions');
|
||
expect(detection).toContain('State that the framework is unknown');
|
||
expect(detection).toContain('If proposing a new framework, settle that choice through Decision procedure in Test step 5');
|
||
expect(detection).toContain('Reuse an exact prior approval; with no selection proposed, ask no framework question');
|
||
expect(detection).toContain('Do not install a framework or write the proposed tests during this review');
|
||
expect(detection).not.toContain('skip test generation');
|
||
expect(shipSkill).toContain('use the bootstrap decision already made in Step 4');
|
||
});
|
||
|
||
test('plan mode adds tests to plan + includes test plan artifact', () => {
|
||
expect(planSkill).toContain('Add missing tests to the plan');
|
||
expect(planSkill).toContain('eng-review-test-plan');
|
||
expect(planSkill).toContain('Test Plan Artifact');
|
||
const artifact = planSkill.split('\n#### Test Plan Artifact\n')[1]!;
|
||
expect(artifact).toContain('After resolving the Test review decisions');
|
||
expect(artifact).toContain('List any unresolved choices separately as pending, not required implementation');
|
||
expect(artifact).toContain('Update this artifact if later approved decisions change the tests');
|
||
expect(artifact).toContain('Use the Review record and write policy above.');
|
||
expect(artifact).toContain('TEST_PLAN_USER=$(whoami)');
|
||
expect(artifact).toContain('sanitized `BRANCH` from gstack-slug');
|
||
expect(artifact).toContain('without an origin, write `local-only`');
|
||
});
|
||
|
||
test('ship mode auto-generates tests + includes before/after count', () => {
|
||
expect(shipSkill).toContain('Generate tests for uncovered paths');
|
||
expect(shipSkill).toContain('Before/after test count');
|
||
expect(shipSkill).toContain('30 code paths max');
|
||
expect(shipSkill).toContain('ship-test-plan');
|
||
});
|
||
|
||
test('review mode uses Fix-First + Review Army for specialist coverage', () => {
|
||
expect(reviewSkill).toContain('Fix-First');
|
||
expect(reviewSkill).toContain('INFORMATIONAL');
|
||
// Review Army handles test coverage via Testing specialist subagent
|
||
expect(reviewSkill).toContain('Review Army');
|
||
expect(reviewSkill).toContain('Testing');
|
||
});
|
||
|
||
test('plan mode does NOT include ship-specific content', () => {
|
||
expect(planSkill).not.toContain('Before/after test count');
|
||
expect(planSkill).not.toContain('30 code paths max');
|
||
expect(planSkill).not.toContain('ship-test-plan');
|
||
});
|
||
|
||
test('review mode does NOT include test plan artifact', () => {
|
||
expect(reviewSkill).not.toContain('Test Plan Artifact');
|
||
expect(reviewSkill).not.toContain('eng-review-test-plan');
|
||
expect(reviewSkill).not.toContain('ship-test-plan');
|
||
});
|
||
|
||
test('review/specialists/ directory has all expected checklist files', () => {
|
||
const specDir = path.join(ROOT, 'review', 'specialists');
|
||
const expected = [
|
||
'testing.md',
|
||
'maintainability.md',
|
||
'security.md',
|
||
'performance.md',
|
||
'data-migration.md',
|
||
'api-contract.md',
|
||
'simplification.md',
|
||
'red-team.md',
|
||
];
|
||
for (const f of expected) {
|
||
expect(fs.existsSync(path.join(specDir, f))).toBe(true);
|
||
}
|
||
});
|
||
|
||
// Regression pins for the simplification specialist (advisory carve-out edits
|
||
// the pre-existing quality_score instruction, so the rendered contract is
|
||
// pinned statically — the carve-out and the early-out line must both survive
|
||
// regeneration verbatim).
|
||
test('simplification advisory carve-out and early-out render into review docs', () => {
|
||
const reviewArmySection = fs.readFileSync(
|
||
path.join(ROOT, 'review', 'sections', 'review-army.md'),
|
||
'utf-8',
|
||
);
|
||
expect(reviewArmySection).toContain('"advisory": true');
|
||
expect(reviewArmySection).toContain('Only specialist findings enter this header and `quality_score`; core findings do not');
|
||
expect(reviewArmySection).toContain('Use the merged NON-advisory specialist findings for both counts and score');
|
||
expect(reviewArmySection).toContain('Simplification: lean already — nothing to cut.');
|
||
expect(reviewArmySection).toContain('net: -N lines possible');
|
||
expect(reviewArmySection).toContain('--simplification');
|
||
// The specialist itself must never carry a verdict-shaped zero-findings line.
|
||
const spec = fs.readFileSync(
|
||
path.join(ROOT, 'review', 'specialists', 'simplification.md'),
|
||
'utf-8',
|
||
);
|
||
expect(spec).toContain('NO FINDINGS');
|
||
expect(spec).not.toContain('Lean already. Ship.');
|
||
// Closed tag vocabulary: the disavowed yagni: frame must not appear.
|
||
expect(spec).toContain('speculative');
|
||
expect(spec.toLowerCase()).not.toContain('"yagni"');
|
||
});
|
||
|
||
test('each specialist file has standard header with scope and output format', () => {
|
||
const specDir = path.join(ROOT, 'review', 'specialists');
|
||
const files = fs.readdirSync(specDir).filter(f => f.endsWith('.md'));
|
||
for (const f of files) {
|
||
const content = fs.readFileSync(path.join(specDir, f), 'utf-8');
|
||
// All specialist files must have Scope and Output/JSON in header
|
||
expect(content).toContain('Scope:');
|
||
expect(content.toLowerCase()).toMatch(/output|json/);
|
||
// Must define NO FINDINGS behavior
|
||
expect(content).toContain('NO FINDINGS');
|
||
}
|
||
});
|
||
|
||
// Regression guard: ship output contains key phrases from before the refactor
|
||
test('ship SKILL.md regression guard — key phrases preserved', () => {
|
||
const regressionPhrases = [
|
||
'Coverage goal: every changed behavior is protected by a test that would catch a real regression.',
|
||
'ASCII coverage diagram',
|
||
'processPayment',
|
||
'refundPayment',
|
||
'billing.test.ts',
|
||
'checkout.e2e.ts',
|
||
'COVERAGE:',
|
||
'QUALITY:',
|
||
'GAPS:',
|
||
'Code paths:',
|
||
'User flows:',
|
||
];
|
||
for (const phrase of regressionPhrases) {
|
||
expect(shipSkill).toContain(phrase);
|
||
}
|
||
});
|
||
|
||
test('ship SKILL.md contains review army specialist dispatch', () => {
|
||
expect(shipSkill).toContain('Specialist Dispatch');
|
||
expect(shipSkill).toContain('Step 9.1');
|
||
expect(shipSkill).toContain('Step 9.2');
|
||
});
|
||
|
||
test('ship SKILL.md contains cross-review finding dedup', () => {
|
||
expect(shipSkill).toContain('Cross-review finding dedup');
|
||
expect(shipSkill).toContain('Step 9.3');
|
||
});
|
||
|
||
test('ship SKILL.md contains re-run idempotency behavior', () => {
|
||
expect(shipSkill).toContain('**Route:**');
|
||
expect(shipSkill.replace(/\s+/g, ' ')).toContain('integrate (1–3) → test and review (4–11.5) → prepare the release (12–15) → verify frozen content (16) → push and publish (17–21)');
|
||
expect(shipSkill.replace(/\s+/g, ' ')).toContain('Every new invocation repeats Steps 1–16, including both reviews and the docs audit');
|
||
expect(shipSkill.replace(/\s+/g, ' ')).toContain('Steps 12, 17 and 19 prevent duplicate bumps, pushes and PRs, never verification');
|
||
});
|
||
});
|
||
|
||
// --- {{TEST_FAILURE_TRIAGE}} resolver tests ---
|
||
|
||
describe('TEST_FAILURE_TRIAGE resolver', () => {
|
||
const shipSkill = readShipUnion();
|
||
|
||
test('contains all 4 triage steps', () => {
|
||
expect(shipSkill).toContain('Step T1: Classify each failure');
|
||
expect(shipSkill).toContain('Step T2: Handle in-branch failures');
|
||
expect(shipSkill).toContain('Step T3: Handle pre-existing failures');
|
||
expect(shipSkill).toContain('Step T4: Execute the chosen action');
|
||
});
|
||
|
||
test('T1 includes classification criteria (in-branch vs pre-existing)', () => {
|
||
expect(shipSkill).toContain('In-branch');
|
||
expect(shipSkill).toContain('Likely pre-existing');
|
||
expect(shipSkill).toContain('git diff origin/');
|
||
});
|
||
|
||
test('T3 branches on REPO_MODE (solo vs collaborative)', () => {
|
||
expect(shipSkill).toContain('REPO_MODE');
|
||
expect(shipSkill).toContain('solo');
|
||
expect(shipSkill).toContain('collaborative');
|
||
});
|
||
|
||
test('solo mode offers fix-now, TODO, and skip options', () => {
|
||
expect(shipSkill).toContain('Investigate and fix now');
|
||
expect(shipSkill).toContain('Add as P0 TODO');
|
||
expect(shipSkill).toContain('Skip');
|
||
});
|
||
|
||
test('collaborative mode offers blame + assign option', () => {
|
||
expect(shipSkill).toContain('Blame + assign GitHub issue');
|
||
expect(shipSkill).toContain('gh issue create');
|
||
});
|
||
|
||
test('defaults ambiguous failures to in-branch (safety)', () => {
|
||
expect(shipSkill).toContain('When ambiguous, default to in-branch');
|
||
});
|
||
});
|
||
|
||
// --- {{PLAN_FILE_REVIEW_REPORT}} resolver tests ---
|
||
|
||
describe('PLAN_FILE_REVIEW_REPORT resolver', () => {
|
||
const REVIEW_SKILLS = ['plan-ceo-review', 'plan-eng-review', 'plan-design-review', 'codex'];
|
||
|
||
for (const [label, skill] of [['Eng', 'plan-eng-review'], ['CEO', 'plan-ceo-review']]) {
|
||
test(`${label} report example is fenced Markdown, and the original escaped fence is rejected`, async () => {
|
||
const { marked } = await import('marked');
|
||
const content = readSkillUnion(skill!);
|
||
const tokens = marked.lexer(content);
|
||
const headings = tokens.filter(token => token.type === 'heading' && token.depth === 2);
|
||
expect(headings.map(token => token.text)).not.toContain('GSTACK REVIEW REPORT');
|
||
const example = tokens.find(token => token.type === 'code'
|
||
&& token.lang === 'markdown' && token.text.startsWith('## GSTACK REVIEW REPORT\n'));
|
||
expect(example).toBeDefined();
|
||
if (example?.type !== 'code') throw new Error('Missing fenced report example');
|
||
expect(example.text).toContain('| Review | Trigger | Why | Runs | Status | Findings |');
|
||
expect(example.text).toContain('`/' + skill + '`');
|
||
// Reproduce the observed literal-backslash delimiters without changing
|
||
// the parser or preprocessing malformed Markdown into valid fences.
|
||
const broken = content.replace(example.raw, example.raw.replaceAll('`', '\\`'));
|
||
expect(broken).not.toBe(content);
|
||
const brokenTokens = marked.lexer(broken);
|
||
expect(brokenTokens.some(token => token.type === 'code' && token.lang === 'markdown'
|
||
&& token.text.startsWith('## GSTACK REVIEW REPORT\n'))).toBe(false);
|
||
expect(brokenTokens.filter(token => token.type === 'heading' && token.depth === 2)
|
||
.map(token => token.text)).toContain('GSTACK REVIEW REPORT');
|
||
});
|
||
}
|
||
|
||
test('Eng renderers use real code delimiters without changing confidence rules or report fields', async () => {
|
||
const {generateConfidenceCalibration} = await import('../scripts/resolvers/confidence');
|
||
const {generateReviewDashboard, generatePlanFileReviewReport, generateCodexPlanReview} = await import('../scripts/resolvers/review');
|
||
const {HOST_PATHS} = await import('../scripts/resolvers/types');
|
||
for (const host of ALL_HOST_CONFIGS) {
|
||
const ctx = {skillName: 'plan-eng-review', tmplPath: 'plan-eng-review/SKILL.md.tmpl',
|
||
host: host.name, paths: HOST_PATHS[host.name]!};
|
||
const confidence = generateConfidenceCalibration(ctx);
|
||
const dashboard = generateReviewDashboard(ctx);
|
||
const report = generatePlanFileReviewReport(ctx);
|
||
const outside = generateCodexPlanReview(ctx);
|
||
// These four resolver fragments contain no intentional escaped-backtick
|
||
// example; the preamble does, and is deliberately outside this check.
|
||
for (const output of [confidence, dashboard, report, outside]) expect(output).not.toContain('\\`');
|
||
expect(confidence).toBe(generateConfidenceCalibration({...ctx, skillName: 'plan-ceo-review'}).replaceAll('\\`', '`'));
|
||
const ceoDashboard = generateReviewDashboard({...ctx, skillName: 'plan-ceo-review'}).replaceAll('\\`', '`');
|
||
const ceoVoiceSource = 'Below the dashboard, group `autoplan-voices` and `design-outside-voices` by workflow run and phase';
|
||
expect(dashboard.replace(/\s+/g, ' ')).toContain(ceoVoiceSource);
|
||
expect(ceoDashboard.replace(/\s+/g, ' ')).toContain(ceoVoiceSource);
|
||
expect(ceoDashboard).toBe(dashboard);
|
||
for (const field of ['status', 'unresolved', 'critical_gaps', 'issues_found', 'mode', 'commit']) {
|
||
expect(report).toContain('`' + field + '`');
|
||
}
|
||
expect(outside).toContain(host.name === 'codex' ? 'claude auth login' : 'codex login');
|
||
expect(outside).toContain('A native result never supplies outside coverage');
|
||
}
|
||
});
|
||
|
||
for (const skill of REVIEW_SKILLS) {
|
||
test(`plan file review report appears in ${skill} generated file`, () => {
|
||
const content = fs.readFileSync(path.join(ROOT, skill, 'SKILL.md'), 'utf-8');
|
||
expect(content).toContain('GSTACK REVIEW REPORT');
|
||
});
|
||
}
|
||
|
||
test('resolver output contains key report elements', () => {
|
||
const content = readSkillUnion('plan-ceo-review'); // carved: report writer moved to section
|
||
expect(content).toContain('Trigger');
|
||
expect(content).toContain('Findings');
|
||
expect(content).toContain('VERDICT');
|
||
expect(content).toContain('/plan-ceo-review');
|
||
expect(content).toContain('/plan-eng-review');
|
||
expect(content).toContain('/plan-design-review');
|
||
expect(content).toContain('/codex review');
|
||
});
|
||
|
||
test('Eng task output follows the same write policy as its report', () => {
|
||
const content = readSkillUnion('plan-eng-review');
|
||
// The task example itself has an H2; bound by the following real section.
|
||
const tasks = content.slice(content.indexOf('## Implementation Tasks'), content.indexOf('## Plan File Review Report'));
|
||
expect(tasks).toContain('only when the Review record and write policy permits it');
|
||
expect(tasks).toContain('label the complete task output not persisted');
|
||
expect(tasks).toContain('write when permitted, including zero tasks');
|
||
expect(tasks).toContain('When writes are permitted and zero tasks were identified');
|
||
expect(tasks).not.toContain('always write');
|
||
});
|
||
});
|
||
|
||
// --- {{PLAN_COMPLETION_AUDIT}} resolver tests ---
|
||
|
||
describe('PLAN_COMPLETION_AUDIT placeholders', () => {
|
||
const shipSkill = readShipUnion();
|
||
const reviewSkill = readSkillUnion('review'); // carved: plan-completion audit moved to sections/plan-completion.md
|
||
|
||
test('ship SKILL.md contains plan completion audit step', () => {
|
||
expect(shipSkill).toContain('Plan Completion Audit');
|
||
expect(shipSkill).toContain('Step 8');
|
||
});
|
||
|
||
test('review SKILL.md contains plan completion in scope drift', () => {
|
||
expect(reviewSkill).toContain('Plan File Discovery');
|
||
expect(reviewSkill).toContain('Actionable Item Extraction');
|
||
expect(reviewSkill).toContain('Integration with Scope Drift Detection');
|
||
});
|
||
|
||
test('both modes share plan file discovery methodology', () => {
|
||
expect(shipSkill).toContain('Plan File Discovery');
|
||
expect(reviewSkill).toContain('Plan File Discovery');
|
||
// Both should have conversation context first
|
||
expect(shipSkill).toContain('Conversation context (primary)');
|
||
expect(reviewSkill).toContain('Conversation context (primary)');
|
||
// Both should have grep fallback
|
||
expect(shipSkill).toContain('Content-based search (fallback)');
|
||
expect(reviewSkill).toContain('Content-based search (fallback)');
|
||
});
|
||
|
||
test('ship mode has gate logic for NOT DONE items', () => {
|
||
expect(shipSkill).toContain('NOT DONE');
|
||
expect(shipSkill).toContain('Stop — implement the missing items');
|
||
expect(shipSkill).toContain('Ship anyway — defer');
|
||
expect(shipSkill).toContain('intentionally dropped');
|
||
});
|
||
|
||
test('review mode is INFORMATIONAL only', () => {
|
||
expect(reviewSkill).toContain('INFORMATIONAL');
|
||
expect(reviewSkill).toContain('MISSING REQUIREMENTS');
|
||
expect(reviewSkill).toContain('SCOPE CREEP');
|
||
});
|
||
|
||
test('item extraction has 50-item cap', () => {
|
||
expect(shipSkill).toContain('at most 50 items');
|
||
});
|
||
|
||
test('uses file-level traceability (not commit-level)', () => {
|
||
expect(shipSkill).toContain('Cite the specific file');
|
||
expect(shipSkill).not.toContain('commit-level traceability');
|
||
});
|
||
});
|
||
|
||
// --- {{PLAN_VERIFICATION_EXEC}} resolver tests ---
|
||
|
||
describe('PLAN_VERIFICATION_EXEC placeholder', () => {
|
||
const shipSkill = readShipUnion();
|
||
|
||
test('ship SKILL.md contains plan verification step', () => {
|
||
expect(shipSkill).toContain('Step 8.1');
|
||
expect(shipSkill).toContain('Plan Verification');
|
||
});
|
||
|
||
test('references the shared explorer without invoking an entire QA workflow', () => {
|
||
const resource = "From the installed /ship SKILL.md's directory, Read `../qa/sections/exploratory.md` in full";
|
||
expect(shipSkill).toContain(resource);
|
||
const load = shipSkill.indexOf(resource);
|
||
const preflight = shipSkill.indexOf('Run the shared preflight;');
|
||
const probes = shipSkill.indexOf('**3. Run smoke and plan checks.**');
|
||
expect(load).toBeGreaterThan(-1);
|
||
expect(preflight).toBeGreaterThan(load);
|
||
expect(probes).toBeGreaterThan(preflight);
|
||
const shared = fs.readFileSync(path.join(ROOT, 'qa/sections/exploratory.md'), 'utf8');
|
||
const selection = shared.indexOf('in full and select the surfaces');
|
||
const methods = shared.indexOf('Read `sections/system-functional.md`');
|
||
expect(selection).toBeGreaterThan(shared.indexOf('Read `sections/scope.md`'));
|
||
expect(methods).toBeGreaterThan(selection);
|
||
expect(shared.indexOf('Write a **charter**')).toBeGreaterThan(methods);
|
||
expect(shipSkill.slice(preflight, probes)).toContain("For browsers, Read QA's `sections/browser-setup.md`");
|
||
expect(shipSkill).toContain('Do not invoke an entire QA skill or start probes here');
|
||
});
|
||
|
||
test('keeps declared browser URLs separate from native functional probes', () => {
|
||
expect(shipSkill).toContain('items use the declared project/plan dev URL');
|
||
expect(shipSkill).toContain('functional items use native tools without discovering a web server');
|
||
expect(shipSkill.replace(/\s+/g, ' ')).toContain('An API URL is not automatically a page');
|
||
});
|
||
|
||
test('retains automatic exploration when there is no plan or verification section', () => {
|
||
expect(shipSkill).toContain('If no verification section or no plan file');
|
||
expect(shipSkill).toContain('Automatic diff-scoped QA still runs');
|
||
expect(shipSkill).toContain('plan checks beyond that smoke budget remain required');
|
||
});
|
||
|
||
test('blocks unavailable required checks instead of silently skipping them', () => {
|
||
const flat = shipSkill.replace(/\s+/g, ' ');
|
||
expect(flat).toContain('Noninteractive runs return blocked');
|
||
expect(flat).toContain("Send failed, blocked or unrun checks through Step 9's required-probe gate, never silently waive them");
|
||
expect(flat).toContain('Missing/unreadable assets block required QA');
|
||
expect(flat).toContain('explicitly accept each named probe\'s concrete risk');
|
||
expect(flat).toContain('Keep actual outcomes and incomplete flags; VERIFY_RESULT stays fail');
|
||
});
|
||
});
|
||
|
||
// --- Coverage gate tests ---
|
||
|
||
describe('Coverage gate in ship', () => {
|
||
const shipSkill = readShipUnion();
|
||
const reviewSkill = readSkillUnion('review'); // carved: testing.md specialist ref lives in sections/review-army.md
|
||
|
||
test('ship SKILL.md contains coverage gate with thresholds', () => {
|
||
expect(shipSkill).toContain('Coverage gate');
|
||
expect(shipSkill).toContain('>= target');
|
||
expect(shipSkill).toContain('< minimum');
|
||
});
|
||
|
||
test('ship SKILL.md supports configurable thresholds via CLAUDE.md', () => {
|
||
expect(shipSkill).toContain('## Test Coverage');
|
||
expect(shipSkill).toContain('Minimum:');
|
||
expect(shipSkill).toContain('Target:');
|
||
});
|
||
|
||
test('coverage gate skips on parse failure (not block)', () => {
|
||
expect(shipSkill).toContain('could not determine percentage — skipping');
|
||
});
|
||
|
||
test('review SKILL.md delegates coverage to Testing specialist', () => {
|
||
// Coverage audit moved to Testing specialist subagent in Review Army
|
||
expect(reviewSkill).toContain('testing.md');
|
||
expect(reviewSkill).toContain('INFORMATIONAL');
|
||
});
|
||
});
|
||
|
||
// --- Ship metrics logging ---
|
||
|
||
describe('Ship metrics logging', () => {
|
||
const shipSkill = readShipUnion();
|
||
|
||
test('ship SKILL.md contains metrics persistence step', () => {
|
||
expect(shipSkill).toContain('Step 20');
|
||
expect(shipSkill).toContain('coverage_pct');
|
||
expect(shipSkill).toContain('plan_items_total');
|
||
expect(shipSkill).toContain('plan_items_done');
|
||
expect(shipSkill).toContain('verification_result');
|
||
});
|
||
});
|
||
|
||
// --- Plan file discovery shared helper ---
|
||
|
||
describe('Plan file discovery shared helper', () => {
|
||
// The shared helper should appear in ship (via PLAN_COMPLETION_AUDIT_SHIP)
|
||
// and in review (via PLAN_COMPLETION_AUDIT_REVIEW)
|
||
const shipSkill = readShipUnion();
|
||
const reviewSkill = readSkillUnion('review'); // carved: plan-completion audit moved to sections/plan-completion.md
|
||
|
||
test('plan file discovery appears in both ship and review', () => {
|
||
expect(shipSkill).toContain('Plan File Discovery');
|
||
expect(reviewSkill).toContain('Plan File Discovery');
|
||
});
|
||
|
||
test('both include conversation context first', () => {
|
||
expect(shipSkill).toContain('Conversation context (primary)');
|
||
expect(reviewSkill).toContain('Conversation context (primary)');
|
||
});
|
||
|
||
test('both include content-based fallback', () => {
|
||
expect(shipSkill).toContain('Content-based search (fallback)');
|
||
expect(reviewSkill).toContain('Content-based search (fallback)');
|
||
});
|
||
});
|
||
|
||
// --- Retro plan completion ---
|
||
|
||
describe('Retro plan completion section', () => {
|
||
// Carved: the narrative report format (incl. Plan Completion) lives in
|
||
// retro/sections/report-format.md — read the skeleton+sections union.
|
||
const retroSkill = readSkillUnion('retro');
|
||
|
||
test('retro SKILL.md contains plan completion section', () => {
|
||
expect(retroSkill).toContain('### Plan Completion');
|
||
expect(retroSkill).toContain('plan_items_total');
|
||
expect(retroSkill).toContain('Plan Completion This Period');
|
||
});
|
||
});
|
||
|
||
// --- Plan status footer in preamble ---
|
||
|
||
describe('Plan status footer in preamble', () => {
|
||
test('preamble contains plan status footer as neutral forward reference to EXIT PLAN MODE GATE', () => {
|
||
// Read any skill that uses PREAMBLE
|
||
const content = readSkillUnion('office-hours'); // carved: Phase 5/6 prose moved to section
|
||
expect(content).toContain('Plan Status Footer');
|
||
expect(content).toContain('GSTACK REVIEW REPORT');
|
||
expect(content).toContain('ExitPlanMode');
|
||
expect(content).toContain('EXIT PLAN MODE GATE');
|
||
// The preamble must NOT impose review-report rules on operational skills
|
||
// that have no review report. It's a forward reference, not enforcement.
|
||
expect(content).not.toContain('NO REVIEWS YET');
|
||
});
|
||
});
|
||
|
||
// --- make-pdf setup ordering ---
|
||
|
||
describe('make-pdf setup ordering', () => {
|
||
test('MAKE-PDF SETUP appears before generic preamble footer sections', () => {
|
||
const content = fs.readFileSync(path.join(ROOT, 'make-pdf', 'SKILL.md'), 'utf-8');
|
||
const preambleIdx = content.indexOf('## Preamble (run first)');
|
||
const setupIdx = content.indexOf('## MAKE-PDF SETUP');
|
||
const planModeIdx = content.indexOf('## Plan Mode Safe Operations');
|
||
const telemetryIdx = content.indexOf('## Telemetry (run last)');
|
||
const workflowIdx = content.indexOf('# make-pdf: publication-quality PDFs from markdown');
|
||
|
||
expect(preambleIdx).toBeGreaterThanOrEqual(0);
|
||
expect(setupIdx).toBeGreaterThan(preambleIdx);
|
||
expect(setupIdx).toBeLessThan(planModeIdx);
|
||
expect(setupIdx).toBeLessThan(telemetryIdx);
|
||
expect(setupIdx).toBeLessThan(workflowIdx);
|
||
expect(content.match(/^## MAKE-PDF SETUP/gm)?.length ?? 0).toBe(1);
|
||
});
|
||
});
|
||
|
||
// --- Skill invocation during plan mode in preamble ---
|
||
|
||
describe('Skill invocation during plan mode in preamble', () => {
|
||
test('preamble contains skill invocation plan mode section', () => {
|
||
const content = readSkillUnion('office-hours'); // carved: Phase 5/6 prose moved to section
|
||
expect(content).toContain('Skill Invocation During Plan Mode');
|
||
expect(content).toContain('precedence over generic plan mode behavior');
|
||
expect(content).toContain('Do not continue the workflow');
|
||
expect(content).toContain('cancel the skill or leave plan mode');
|
||
});
|
||
});
|
||
|
||
// --- {{SPEC_REVIEW_LOOP}} resolver tests ---
|
||
|
||
describe('SPEC_REVIEW_LOOP resolver', () => {
|
||
const content = readSkillUnion('office-hours'); // carved: Phase 5/6 prose moved to section
|
||
const { generateSpecReviewLoop } = require('../scripts/resolvers/review');
|
||
const { HOST_PATHS } = require('../scripts/resolvers/types');
|
||
const render = (skillName: string, host = 'claude') => generateSpecReviewLoop({
|
||
skillName,
|
||
tmplPath: `${skillName}/SKILL.md.tmpl`,
|
||
host,
|
||
paths: HOST_PATHS[host],
|
||
});
|
||
|
||
test('office-hours prepares a complete reviewer prompt on every host', () => {
|
||
const { renderOfficeHoursReviewerPrompt } = require('../lib/office-hours-review');
|
||
for (const host of Object.keys(HOST_PATHS)) {
|
||
const output = render('office-hours', host);
|
||
expect(output).toContain('gstack-office-hours-review prepare --design');
|
||
expect(output).toContain('Before EACH dispatch');
|
||
expect(output).toContain('all preceding valid round files in order');
|
||
expect(output).toContain('string unchanged as the prompt');
|
||
expect(output).toContain('Read the entire prepared prompt');
|
||
expect(output).toContain('A parent Read does not deliver the file to the reviewer');
|
||
const prompt = renderOfficeHoursReviewerPrompt({ document: '/tmp/design.md', verdictPath: '/tmp/round-1.json' });
|
||
expect(prompt).toContain('design and coaching document, produced before engineering');
|
||
expect(prompt).toContain("startup-mode 'The Assignment'");
|
||
expect(prompt).toContain("both modes' 'What I noticed about how you think'");
|
||
expect(prompt).toContain('evaluate their evidence and usefulness');
|
||
expect(prompt).toContain('do not remove them merely because they are coaching content');
|
||
expect(prompt).toContain('Unknown customer facts may remain explicit Open Questions or assignments; do not invent answers');
|
||
expect(prompt).toContain('unsupported claims, contradictions, safety/correctness risks');
|
||
expect(prompt).toContain('missing behavior needed by the approach the document actually commits to');
|
||
expect(prompt).toContain('a contradiction or a required behavior an open question does not resolve it');
|
||
expect(prompt).toContain('clear enough for user approval and the next engineering review');
|
||
expect(prompt).toContain('Are open discovery questions distinguished from committed behavior?');
|
||
expect(prompt).toContain('Flag ambiguous or missing behavior in the chosen approach');
|
||
expect(prompt).not.toContain('Could an engineer implement this without asking questions?');
|
||
expect(prompt).toContain('return that identical JSON as your entire response');
|
||
expect(prompt).toContain('include every unresolved problem and necessary remedy, including minor findings');
|
||
expect(prompt).toContain('classify EVERY preceding finding as resolved, persisting, or unverified');
|
||
expect(prompt).toContain('Absence from the new findings list is not confirmation');
|
||
expect(prompt).toContain('specific document decision/behavior proving the status');
|
||
expect(prompt).toContain('Never invent customer answers');
|
||
expect(prompt).toContain('"version": 1');
|
||
expect(prompt).toContain('"prior": []');
|
||
}
|
||
});
|
||
|
||
test('office-hours checks stop conditions before edits and preserves concerns mechanically', () => {
|
||
const output = render('office-hours');
|
||
const step2 = output.slice(output.indexOf('**Step 2:'), output.indexOf('**Step 3:'));
|
||
expect(step2).toContain('BEFORE fixing any findings or dispatching again');
|
||
expect(step2).toContain('gstack-office-hours-review check');
|
||
expect(step2).toContain('PASS: no unresolved findings');
|
||
expect(step2).toContain('CONVERGENCE: the reviewer explicitly marked a prior obligation persisting');
|
||
expect(step2).toContain('concrete prior/current finding pair and document evidence');
|
||
expect(step2).toContain('Shared topic labels or new refinements alone are insufficient');
|
||
expect(step2).toContain('MAX_ITERATIONS: round 3 completed; stop');
|
||
expect(step2).toContain('CONTINUE: fix the listed findings');
|
||
expect(step2).toContain('On a stop, do not fix again or re-dispatch');
|
||
expect(step2).toContain('finalizer before approval');
|
||
expect(step2).toContain('gstack-office-hours-review finalize --design');
|
||
expect(step2).toContain('Recording concerns does not mark them fixed');
|
||
expect(step2).toContain('Do not edit that generated section');
|
||
expect(step2).toContain('Then proceed to Step 3 and the existing user approval');
|
||
expect(step2.indexOf('gstack-office-hours-review check')).toBeLessThan(step2.indexOf('CONTINUE: fix'));
|
||
});
|
||
|
||
test('office-hours finalizes the report after writing it and labels derivable metrics', () => {
|
||
const output = render('office-hours');
|
||
expect(output).toContain('with `--report "<report-path>"` after the report exists');
|
||
expect(output).toContain('Disposition mechanically');
|
||
expect(output).toContain('Do not\nsummarize or replace that section afterward');
|
||
expect(output).toContain('Preserve the Assignment, coaching, approval, and Handoff');
|
||
expect(output).toContain('Confirmed resolutions require explicit later reviewer evidence');
|
||
expect(output).toContain('finding observations across rounds');
|
||
expect(output).toContain('attempted fix\nrounds are counted separately');
|
||
expect(output).toContain('An unavailable score is null, never invented');
|
||
expect(output).not.toContain('M issues caught and fixed');
|
||
});
|
||
|
||
test('office-hours errors stay explicit and preserve preceding valid evidence', () => {
|
||
const output = render('office-hours');
|
||
expect(output).toContain('A missing or invalid verdict is an explicit review failure, never PASS');
|
||
expect(output).toContain('Preserve the\nfailed output and its error');
|
||
expect(output).toContain('--unreviewed');
|
||
expect(output).toContain('preceding valid round files');
|
||
expect(output).toContain('Do not fabricate JSON or hide a completed verdict');
|
||
expect(output).toContain('quality bonus, not an approval gate');
|
||
});
|
||
|
||
test('CEO keeps its loop limits, scoring, failure handling, and reporting', () => {
|
||
const ceo = render('plan-ceo-review');
|
||
const report = ceo.split('**Step 3:')[1]!;
|
||
expect(report).toContain('For an unavailable review or missing/invalid grade, use JSON `null`');
|
||
expect(ceo).toContain('implementable without follow-up questions');
|
||
expect(ceo).not.toContain('design and coaching document');
|
||
expect(ceo).not.toContain('gstack-office-hours-review');
|
||
const dispatch = ceo.slice(ceo.indexOf('**Step 1:'), ceo.indexOf('**Step 2:'));
|
||
expect(dispatch).toContain("Read Agent's tool definition");
|
||
expect(dispatch).toContain('Set `run_in_background: false` if that field is available; omit it otherwise');
|
||
expect(dispatch).toContain('If the result contains a completed review, consume it');
|
||
expect(dispatch).toContain("If it returns a pending task, use the host's wait tool");
|
||
expect(dispatch).toContain('With no wait tool, end this response and resume on its completion notification');
|
||
expect(dispatch).toContain('While waiting, do not advance, edit either input or launch another reviewer');
|
||
expect(dispatch).toContain('both complete labeled texts');
|
||
expect(dispatch).toContain('all five dimensions');
|
||
const fixes = ceo.slice(ceo.indexOf('**Step 2:'), ceo.indexOf('**Step 3:'));
|
||
const stages = ['use 0D for new or reopened choices', 'amend the working plan',
|
||
're-dispatch with both updated inputs and the same instructions'].map(text => fixes.indexOf(text));
|
||
expect(stages.every(index => index >= 0)).toBe(true);
|
||
expect(stages).toEqual([...stages].sort((a, b) => a - b));
|
||
expect(fixes).toContain('Keep both consistent');
|
||
expect(fixes).toContain('Make at most three reviewer launches');
|
||
expect(fixes).toContain('Stop after the third review, or when consecutive reviews repeat the same unresolved issues');
|
||
expect(fixes).toContain('If launch or review fails, times out, or cannot review both complete inputs, stop the loop');
|
||
expect(fixes).toContain('Preserve the failure and all prior findings');
|
||
expect(fixes).toContain('Continue to Step 3 to record the unavailable outcome; a successful reviewer result is not required');
|
||
expect(fixes).toContain('A missing score alone does not require another review');
|
||
expect(ceo).toContain('Spec review unavailable — presenting unreviewed doc.');
|
||
expect(report).toContain('List unresolved issues under "## Reviewer Concerns"');
|
||
expect(report).toContain('citing the owning input');
|
||
expect(report).toContain("SCORE is the latest attempt's reported 1–10 grade after reviewing both full inputs");
|
||
expect(report).toContain('Label earlier grades "prior review score"');
|
||
expect(report).toContain('reviewer-confirmed fixes');
|
||
expect(report).toContain('Use actual counts, never estimates');
|
||
expect(report.replace(/\s+/g, ' ')).toContain('When writing is forbidden, show the actual fields as not persisted and continue without writing');
|
||
expect(report).toContain('failed mkdir or append stops the review');
|
||
expect(report.replace(/\s+/g, ' ')).toContain('Recording the **0H spec-review metrics** is required when writing is permitted, even if the reviewer failed');
|
||
expect(report.replace(/\s+/g, ' ')).toContain('If the reviewer fails, report that limit and continue after recording the outcome; if a required save fails, stop before claiming completion');
|
||
expect(report).toContain('mkdir -p ~/.gstack/analytics || exit 1');
|
||
expect(report).toContain('>> ~/.gstack/analytics/spec-review.jsonl || exit 1');
|
||
expect(report).not.toContain('Your doc survived');
|
||
});
|
||
|
||
test('CEO spec report separates findings, confirmed fixes, and unresolved concerns', () => {
|
||
const report = render('plan-ceo-review').split('**Step 3: Report and persist metrics**')[1]!.replace(/\s+/g, ' ');
|
||
expect(report).toContain('ITERATIONS counts actual reviewer launches');
|
||
expect(report).toContain('FOUND, FIXED and REMAINING count reported issues, reviewer-confirmed fixes and reported unresolved issues');
|
||
expect(report).toContain('Use actual counts, never estimates');
|
||
expect(report).not.toContain('M issues caught and fixed');
|
||
expect(report).toContain('List unresolved issues under "## Reviewer Concerns"');
|
||
});
|
||
|
||
test('CEO spec review receives the full source plan as well as the scope artifact on every host', () => {
|
||
for (const host of Object.keys(HOST_PATHS)) {
|
||
const output = render('plan-ceo-review', host).replace(/\s+/g, ' ');
|
||
expect(output).toContain('Both saved absolute paths, or both complete labeled texts if either input is not persisted: CEO scope summary and current amended working plan');
|
||
expect(output).toContain('Read both inputs in full');
|
||
expect(output).toContain('Evaluate them together on all five dimensions');
|
||
expect(output).toContain('Flag contradictions, unsupported accepted expansions and required behavior missing from both');
|
||
expect(output).toContain('report that failure instead of grading partial input');
|
||
}
|
||
const source = fs.readFileSync(path.join(ROOT, 'plan-ceo-review', 'SKILL.md.tmpl'), 'utf8');
|
||
expect(source).toContain('## Plan under review\n{working plan path, or');
|
||
const template = source.replace(/\s+/g, ' ');
|
||
expect(template).toContain('Prepare the full amended working plan and a separate, consistent CEO scope summary');
|
||
expect(template).toContain('the summary cannot serve as the plan');
|
||
expect(template).toContain('**Save or present both inputs under the storage policy.**');
|
||
});
|
||
|
||
test('CEO shares both inputs after spec review and owns unresolved concerns in its scope document', () => {
|
||
const template = fs.readFileSync(path.join(ROOT, 'plan-ceo-review', 'SKILL.md.tmpl'), 'utf8');
|
||
const persist = (template.split('### 0H.')[1]?.split('### 0I.')[0] ?? '').replace(/\s+/g, ' ');
|
||
const review = persist.indexOf('{{SPEC_REVIEW_LOOP}}');
|
||
const sharing = persist.indexOf('present both inputs for final scope-document approval');
|
||
expect(review).toBeGreaterThanOrEqual(0);
|
||
expect(sharing).toBeGreaterThan(review);
|
||
const output = render('plan-ceo-review');
|
||
const processing = output.split('**Step 2:')[1]!.split('**Step 3:')[0]!;
|
||
const report = output.split('**Step 3:')[1] ?? '';
|
||
expect(processing).toContain('consecutive reviews repeat the same unresolved issues');
|
||
expect(processing).toContain('Preserve the failure and all prior findings');
|
||
expect(report).toContain('CEO summary');
|
||
expect(report).toContain('Reviewer Concerns');
|
||
expect(report).toMatch(/owning\s+input/);
|
||
});
|
||
|
||
test('CEO reviewer receives one complete prompt without duplicate scoring instructions', () => {
|
||
const output = render('plan-ceo-review');
|
||
const dispatch = output.split('**Step 1: Dispatch reviewer subagent**')[1]?.split('**Step 2:')[0] ?? '';
|
||
expect(dispatch.match(/Read both inputs in full/g)).toHaveLength(1);
|
||
expect(dispatch).not.toContain('Read these documents and review them');
|
||
expect(dispatch.match(/quality score/g)).toHaveLength(1);
|
||
expect(dispatch).toContain('A quality score (1-10) across all dimensions');
|
||
expect(dispatch).toContain('Overall PASS only if all dimensions pass');
|
||
expect(dispatch).toContain('For each dimension, PASS or numbered issues with suggested fixes');
|
||
expect(dispatch).toContain('Cite input and requirement for each finding');
|
||
expect(dispatch).toContain('report that failure instead of grading');
|
||
});
|
||
|
||
test('contains all 5 review dimensions', () => {
|
||
for (const dim of ['Completeness', 'Consistency', 'Clarity', 'Scope', 'Feasibility']) {
|
||
expect(content).toContain(dim);
|
||
}
|
||
});
|
||
|
||
test('references Agent tool for subagent dispatch', () => {
|
||
expect(content).toMatch(/Agent.*tool/i);
|
||
});
|
||
|
||
test('specifies max 3 iterations', () => {
|
||
expect(content).toMatch(/3.*iteration|maximum.*3/i);
|
||
});
|
||
|
||
test('includes quality score', () => {
|
||
expect(content).toContain('quality score');
|
||
});
|
||
|
||
test('includes metrics path', () => {
|
||
expect(content).toContain('spec-review.jsonl');
|
||
});
|
||
|
||
test('includes convergence guard', () => {
|
||
expect(content).toMatch(/[Cc]onvergence/);
|
||
});
|
||
|
||
test('includes graceful failure handling', () => {
|
||
expect(content).toMatch(/skip.*review|unavailable/i);
|
||
});
|
||
});
|
||
|
||
// --- {{DESIGN_SKETCH}} resolver tests ---
|
||
|
||
describe('DESIGN_SKETCH resolver', () => {
|
||
const content = readSkillUnion('office-hours'); // carved: Phase 5/6 prose moved to section
|
||
|
||
test('references DESIGN.md for design system constraints', () => {
|
||
expect(content).toContain('DESIGN.md');
|
||
});
|
||
|
||
test('contains wireframe or sketch terminology', () => {
|
||
expect(content).toMatch(/wireframe|sketch/i);
|
||
});
|
||
|
||
test('wireframes render through gstack-render (Aside first)', () => {
|
||
expect(content).toContain('gstack-render.ts');
|
||
});
|
||
|
||
test('references screenshot capture', () => {
|
||
expect(content).toContain('--screenshot');
|
||
});
|
||
|
||
test('specifies rough aesthetic', () => {
|
||
expect(content).toMatch(/[Rr]ough|hand-drawn/);
|
||
});
|
||
|
||
test('includes skip conditions', () => {
|
||
expect(content).toMatch(/no UI component|skip/i);
|
||
});
|
||
});
|
||
|
||
// --- {{CODEX_SECOND_OPINION}} resolver tests ---
|
||
|
||
describe('CODEX_SECOND_OPINION resolver', () => {
|
||
const content = readSkillUnion('office-hours'); // carved: Phase 5/6 prose moved to section
|
||
const codexContent = readExternalSkillUnion(EXTERNAL_OUT, '.agents', 'office-hours');
|
||
|
||
test('Phase 3.5 section appears in office-hours SKILL.md', () => {
|
||
expect(content).toContain('Phase 3.5: Cross-Model Second Opinion');
|
||
});
|
||
|
||
test('contains codex exec invocation', () => {
|
||
expect(content).toContain('codex exec');
|
||
});
|
||
|
||
test('contains opt-in AskUserQuestion text', () => {
|
||
expect(content).toContain('second opinion from an independent AI perspective');
|
||
});
|
||
|
||
test('contains cross-model synthesis instructions', () => {
|
||
expect(content).toMatch(/[Ss]ynthesis/);
|
||
expect(content).toContain('Where Claude agrees with the second opinion');
|
||
});
|
||
|
||
test('contains Claude subagent fallback', () => {
|
||
expect(content).toContain('CODEX_MODE: not_installed');
|
||
expect(content).toContain('If preflight is not ready');
|
||
expect(content).toContain('Agent tool');
|
||
expect(content).toContain('SECOND OPINION (Claude subagent)');
|
||
});
|
||
|
||
test('contains premise revision check', () => {
|
||
expect(content).toContain('Codex challenged premise');
|
||
});
|
||
|
||
test('contains error handling for auth, timeout, and empty', () => {
|
||
expect(content).toMatch(/[Aa]uth.*fail/);
|
||
expect(content).toMatch(/[Tt]imeout/);
|
||
expect(content).toMatch(/[Ee]mpty response/);
|
||
});
|
||
|
||
test('Codex host runs Phase 3.5 through Claude Code', () => {
|
||
expect(codexContent).toContain('Phase 3.5: Cross-Model Second Opinion');
|
||
expect(codexContent).toContain('gstack-claude-code');
|
||
expect(codexContent).toContain('Claude Code');
|
||
expect(codexContent).not.toMatch(/codex\s+(?:exec|review)\s/);
|
||
});
|
||
});
|
||
|
||
// --- Codex filesystem boundary tests ---
|
||
|
||
describe('Codex filesystem boundary', () => {
|
||
// Skills that call codex exec/review and should contain boundary text
|
||
const CODEX_CALLING_SKILLS = [
|
||
'codex', // /codex skill — 3 modes
|
||
'autoplan', // /autoplan — CEO/design/eng voices
|
||
'review', // /review — adversarial step resolver
|
||
'ship', // /ship — adversarial step resolver
|
||
'plan-eng-review', // outside voice resolver
|
||
'plan-ceo-review', // outside voice resolver
|
||
'office-hours', // second opinion resolver
|
||
];
|
||
|
||
const BOUNDARY_MARKER = 'Do NOT read or execute any';
|
||
|
||
test('boundary instruction appears in all skills that call codex', () => {
|
||
for (const skill of CODEX_CALLING_SKILLS) {
|
||
// Union: ship's codex call lives in sections/adversarial.md after the carve.
|
||
const content = readSkillUnion(skill);
|
||
expect(content).toContain(BOUNDARY_MARKER);
|
||
}
|
||
});
|
||
|
||
test('codex skill has Filesystem Boundary section', () => {
|
||
const content = fs.readFileSync(path.join(ROOT, 'codex', 'SKILL.md'), 'utf-8');
|
||
expect(content).toContain('## Filesystem Boundary');
|
||
expect(content).toContain('skill definitions meant for a different AI system');
|
||
});
|
||
|
||
test('codex skill has rabbit-hole detection rule', () => {
|
||
const content = fs.readFileSync(path.join(ROOT, 'codex', 'SKILL.md'), 'utf-8');
|
||
expect(content).toContain('Detect skill-file rabbit holes');
|
||
expect(content).toContain('gstack-update-check');
|
||
expect(content).toContain('Consider retrying');
|
||
});
|
||
|
||
test('review.ts CODEX_BOUNDARY constant is interpolated into resolver output', () => {
|
||
// The adversarial step resolver should include boundary text in codex exec
|
||
// prompts. Carved: the adversarial step lives in sections/adversarial.md.
|
||
const reviewContent = readSkillUnion('review');
|
||
// Boundary should appear near codex exec invocations
|
||
const boundaryIdx = reviewContent.indexOf(BOUNDARY_MARKER);
|
||
const codexExecIdx = reviewContent.indexOf('codex exec');
|
||
// Both must exist and boundary must come before a codex exec call
|
||
expect(boundaryIdx).toBeGreaterThan(-1);
|
||
expect(codexExecIdx).toBeGreaterThan(-1);
|
||
});
|
||
|
||
test('autoplan boundary text avoids host-specific paths for cross-host compatibility', () => {
|
||
const content = fs.readFileSync(path.join(ROOT, 'autoplan', 'SKILL.md.tmpl'), 'utf-8');
|
||
// autoplan template uses generic 'skills/gstack' pattern instead of host-specific
|
||
// paths like ~/.claude/ or .agents/skills (which break Codex/Claude output tests)
|
||
const boundaryStart = content.indexOf('Filesystem Boundary');
|
||
const boundaryEnd = content.indexOf('---', boundaryStart + 1);
|
||
const boundarySection = content.slice(boundaryStart, boundaryEnd);
|
||
expect(boundarySection).not.toContain('~/.claude/');
|
||
expect(boundarySection).not.toContain('.agents/skills');
|
||
expect(boundarySection).toContain('skills/gstack');
|
||
expect(boundarySection).toContain(BOUNDARY_MARKER);
|
||
});
|
||
});
|
||
|
||
// --- {{BENEFITS_FROM}} resolver tests ---
|
||
|
||
describe('BENEFITS_FROM resolver', () => {
|
||
const ceoContent = fs.readFileSync(path.join(ROOT, 'plan-ceo-review', 'SKILL.md'), 'utf-8');
|
||
const engContent = readSkillUnion('plan-eng-review'); // carved
|
||
|
||
test('plan-ceo-review contains prerequisite skill offer', () => {
|
||
expect(ceoContent).toContain('Prerequisite Skill Offer');
|
||
expect(ceoContent).toContain('/office-hours');
|
||
});
|
||
|
||
test('plan-eng-review contains prerequisite skill offer', () => {
|
||
expect(engContent).toContain('Prerequisite Skill Offer');
|
||
expect(engContent).toContain('/office-hours');
|
||
});
|
||
|
||
test('the optional Eng prerequisite ends before mandatory engineering review', () => {
|
||
const offer = extractMarkdownSection(engContent, '## Prerequisite Skill Offer');
|
||
expect(offer).toContain('Build the next full decision brief from these facts and options, using the preamble transport, numbering and format');
|
||
expect(offer).toContain('B) Skip — proceed with standard review');
|
||
expect(offer).toContain('Then proceed normally. Do not re-offer later in the session');
|
||
expect(offer).not.toContain('### Step 0: Scope Challenge');
|
||
const review = extractMarkdownSection(engContent, '## Engineering review');
|
||
expect(review).toContain('### Step 0: Scope Challenge');
|
||
expect(review).toContain('Scope Challenge is mandatory before Section 1');
|
||
expect(review).toContain('sections/review-sections.md');
|
||
});
|
||
|
||
test('Eng initial and prerequisite recheck both discover the canonical override and fall back on slug failure', () => {
|
||
const initial = extractMarkdownSection(engContent, '### Design Doc Check').match(/```bash\n([\s\S]*?)\n```/)![1]!;
|
||
const offer = extractMarkdownSection(engContent, '## Prerequisite Skill Offer');
|
||
expect(offer).toContain('After /office-hours completes, rerun the complete **Design Doc Check** block above');
|
||
expect(offer).toContain('This is a fresh execution');
|
||
expect(offer).toContain('Do not rerun the preamble or re-offer the prerequisite');
|
||
expect(offer).not.toContain('_REVIEW_SLUG=');
|
||
const recheck = initial; // Execute the one canonical block again after the prerequisite.
|
||
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'eng-design-slug-'));
|
||
try {
|
||
const home = path.join(dir, 'home'), cwd = path.join(dir, 'project');
|
||
const helper = path.join(home, '.claude/skills/gstack/bin/gstack-slug');
|
||
fs.mkdirSync(path.dirname(helper), {recursive: true});
|
||
fs.mkdirSync(cwd);
|
||
fs.copyFileSync(path.join(ROOT, 'bin/gstack-slug'), helper);
|
||
fs.chmodSync(helper, 0o755);
|
||
const expected = path.join(home, '.gstack/projects/canonical-override/session-unknown-design-current.md');
|
||
const wrong = path.join(home, '.gstack/projects/project/session-unknown-design-wrong.md');
|
||
for (const file of [expected, wrong]) {
|
||
fs.mkdirSync(path.dirname(file), {recursive: true});
|
||
fs.writeFileSync(file, '# Design fixture\n');
|
||
}
|
||
const env = {...process.env, HOME: home, GSTACK_HOME: path.join(home, '.gstack'),
|
||
GSTACK_PROJECT_SLUG: 'canonical-override', GIT_CEILING_DIRECTORIES: dir};
|
||
for (const command of [initial, recheck]) {
|
||
expect(command).toContain('if _REVIEW_SLUG=$(~/.claude/skills/gstack/bin/gstack-slug); then');
|
||
expect(command).toContain('eval "$_REVIEW_SLUG"');
|
||
expect(command).toContain('echo "No design doc found"');
|
||
expect(command).not.toContain('remote-slug');
|
||
const result = runCapturedCommand('bash', ['-c', command], {cwd, env, timeout: 5000, captureStdout: true});
|
||
expect(result.status, result.stderr).toBe(0);
|
||
expect(result.stdout).toContain(`Design doc found: ${expected}`);
|
||
expect(result.stdout).not.toContain(wrong);
|
||
}
|
||
fs.writeFileSync(helper, '#!/usr/bin/env bash\necho "slug fixture failed" >&2\nexit 23\n', {mode: 0o755});
|
||
for (const command of [initial, recheck]) {
|
||
const result = runCapturedCommand('bash', ['-c', command], {cwd, env, timeout: 5000, captureStdout: true});
|
||
expect(result.status).toBe(0);
|
||
expect(result.stderr).toContain('slug fixture failed');
|
||
expect(result.stdout).not.toContain('Design doc found:');
|
||
expect(result.stdout).toContain('No design doc found');
|
||
}
|
||
} finally { fs.rmSync(dir, {recursive: true, force: true}); }
|
||
});
|
||
|
||
test('offer includes graceful decline', () => {
|
||
expect(ceoContent).toContain('No worries');
|
||
});
|
||
|
||
test('skills without benefits-from do NOT have prerequisite offer', () => {
|
||
const qaContent = fs.readFileSync(path.join(ROOT, 'qa', 'SKILL.md'), 'utf-8');
|
||
expect(qaContent).not.toContain('Prerequisite Skill Offer');
|
||
});
|
||
|
||
test('inline invocation — no "another window" language', () => {
|
||
expect(ceoContent).not.toContain('another window');
|
||
expect(engContent).not.toContain('another window');
|
||
});
|
||
|
||
test('inline invocation — read-and-follow path present', () => {
|
||
expect(ceoContent).toContain('office-hours/SKILL.md');
|
||
expect(engContent).toContain('office-hours/SKILL.md');
|
||
});
|
||
|
||
test('BENEFITS_FROM delegates to INVOKE_SKILL pattern', () => {
|
||
// Should contain the INVOKE_SKILL-style loading prose (not the old manual skip list)
|
||
expect(engContent).toContain('Follow its instructions from top to bottom');
|
||
expect(engContent).toContain('skipping these sections');
|
||
expect(ceoContent).toContain('Follow its instructions from top to bottom');
|
||
});
|
||
});
|
||
|
||
// --- {{INVOKE_SKILL}} resolver tests ---
|
||
|
||
describe('INVOKE_SKILL resolver', () => {
|
||
const ceoContent = fs.readFileSync(path.join(ROOT, 'plan-ceo-review', 'SKILL.md'), 'utf-8');
|
||
|
||
test('plan-ceo-review uses INVOKE_SKILL for mid-session office-hours fallback', () => {
|
||
// The mid-session detection path should use INVOKE_SKILL-generated prose
|
||
expect(ceoContent).toContain('office-hours/SKILL.md');
|
||
expect(ceoContent).toContain('Follow its instructions from top to bottom');
|
||
});
|
||
|
||
test('INVOKE_SKILL output includes default skip list', () => {
|
||
expect(ceoContent).toContain('Preamble (run first)');
|
||
expect(ceoContent).toContain('Telemetry (run last)');
|
||
expect(ceoContent).toContain('AskUserQuestion Format');
|
||
});
|
||
|
||
test('INVOKE_SKILL output includes error handling', () => {
|
||
expect(ceoContent).toContain('If unreadable');
|
||
expect(ceoContent).toContain('Could not load');
|
||
});
|
||
|
||
test('template uses {{INVOKE_SKILL:office-hours}} placeholder', () => {
|
||
const tmpl = fs.readFileSync(path.join(ROOT, 'plan-ceo-review', 'SKILL.md.tmpl'), 'utf-8');
|
||
expect(tmpl).toContain('{{INVOKE_SKILL:office-hours}}');
|
||
});
|
||
});
|
||
|
||
// --- {{CHANGELOG_WORKFLOW}} resolver tests ---
|
||
|
||
describe('CHANGELOG_WORKFLOW resolver', () => {
|
||
const shipContent = readShipUnion();
|
||
|
||
test('ship SKILL.md contains changelog workflow', () => {
|
||
expect(shipContent).toContain('CHANGELOG (auto-generate)');
|
||
expect(shipContent).toContain('git log origin/<base>..HEAD --oneline');
|
||
});
|
||
|
||
test('changelog workflow includes cross-check step', () => {
|
||
expect(shipContent).toContain('Cross-check');
|
||
expect(shipContent).toContain('Every commit must map to at least one bullet point');
|
||
});
|
||
|
||
test('changelog workflow includes voice guidance', () => {
|
||
expect(shipContent).toContain('Lead with what the user can now **do**');
|
||
});
|
||
|
||
test('template uses {{CHANGELOG_WORKFLOW}} placeholder', () => {
|
||
// Post-carve (T9): the skeleton points to the changelog section, which carries
|
||
// the resolver. Neither should inline the old changelog content.
|
||
const skel = fs.readFileSync(path.join(ROOT, 'ship', 'SKILL.md.tmpl'), 'utf-8');
|
||
const changelogSection = fs.readFileSync(path.join(ROOT, 'ship', 'sections', 'changelog.md.tmpl'), 'utf-8');
|
||
expect(skel).toContain('{{SECTION:changelog}}');
|
||
expect(changelogSection).toContain('{{CHANGELOG_WORKFLOW}}');
|
||
expect(skel + changelogSection).not.toContain('Group commits by theme');
|
||
});
|
||
|
||
test('changelog workflow includes keep-changelog format', () => {
|
||
expect(shipContent).toContain('### Added');
|
||
expect(shipContent).toContain('### Fixed');
|
||
});
|
||
});
|
||
|
||
// --- Duplicate {{PREAMBLE}} guard (#2508/#2362) ---
|
||
|
||
describe('assertSinglePreamble', () => {
|
||
test('one {{PREAMBLE}} passes', () => {
|
||
expect(() => assertSinglePreamble('a\n{{PREAMBLE}}\nb', 'x/SKILL.md.tmpl')).not.toThrow();
|
||
});
|
||
|
||
test('zero {{PREAMBLE}} passes (sections have none)', () => {
|
||
expect(() => assertSinglePreamble('no macro here', 'x/sections/y.md.tmpl')).not.toThrow();
|
||
});
|
||
|
||
test('a second occurrence throws with the template path — even in prose', () => {
|
||
// The original #2508 bug WAS a prose mention: "emitted by {{PREAMBLE}}'s
|
||
// preamble bash". Resolution is context-blind, so the guard must be too.
|
||
const tmpl = '{{PREAMBLE}}\n\n...later: emitted by {{PREAMBLE}}\'s preamble bash';
|
||
expect(() => assertSinglePreamble(tmpl, 'spec/SKILL.md.tmpl')).toThrow(/spec\/SKILL\.md\.tmpl.*2 times/);
|
||
});
|
||
});
|
||
|
||
// --- Parameterized resolver infrastructure tests ---
|
||
|
||
describe('parameterized resolver support', () => {
|
||
test('gen-skill-docs regex handles colon-separated args', () => {
|
||
// Verify the template containing {{INVOKE_SKILL:office-hours}} was processed
|
||
// without leaving unresolved placeholders
|
||
const ceoContent = fs.readFileSync(path.join(ROOT, 'plan-ceo-review', 'SKILL.md'), 'utf-8');
|
||
expect(ceoContent).not.toMatch(/\{\{INVOKE_SKILL:[^}]+\}\}/);
|
||
});
|
||
|
||
test('templates with parameterized resolvers pass unresolved check', () => {
|
||
// All generated SKILL.md files should have no unresolved {{...}} placeholders
|
||
const skillDirs = fs.readdirSync(ROOT).filter(d =>
|
||
fs.existsSync(path.join(ROOT, d, 'SKILL.md'))
|
||
);
|
||
for (const dir of skillDirs) {
|
||
const content = fs.readFileSync(path.join(ROOT, dir, 'SKILL.md'), 'utf-8');
|
||
const unresolved = content.match(/\{\{[A-Z_]+(?::[^}]*)?\}\}/g);
|
||
if (unresolved) {
|
||
throw new Error(`${dir}/SKILL.md has unresolved placeholders: ${unresolved.join(', ')}`);
|
||
}
|
||
}
|
||
});
|
||
});
|
||
|
||
// --- Preamble routing injection tests ---
|
||
|
||
describe('preamble routing injection (bin/gstack-skill-start emission layer)', () => {
|
||
// Token-reduction Phase 2: the routing-injection prose left the rendered
|
||
// preamble entirely — bin/gstack-skill-start probes, gates, and emits the
|
||
// whole flow as a GSTACK_INSTRUCTION block (with the AUQ, the routing rules
|
||
// to append, and the decline ack all INSIDE the block). Absence from the
|
||
// renders is pinned by test/onboarding-moved-literals.test.ts (tombstone);
|
||
// this suite pins the gate structure and the emitted block's content.
|
||
const routingBlock = (() => {
|
||
const start = SKILL_START_SCRIPT.indexOf('_emit_block routing-injection');
|
||
expect(start).toBeGreaterThan(0);
|
||
return SKILL_START_SCRIPT.slice(start, SKILL_START_SCRIPT.indexOf('\nEOI', start));
|
||
})();
|
||
|
||
test('routing probe checks CLAUDE.md and AGENTS.md (now in gstack-skill-start)', () => {
|
||
// #2500: the probe iterates CLAUDE.md AND AGENTS.md — non-Claude hosts
|
||
// route skills via AGENTS.md, the cross-harness convention file.
|
||
expect(SKILL_START_SCRIPT).toContain('for _RF in CLAUDE.md AGENTS.md');
|
||
expect(SKILL_START_SCRIPT).toContain('grep -q "## Skill routing" "$_RF"');
|
||
expect(SKILL_START_SCRIPT).toContain('echo "HAS_ROUTING: $_HAS_ROUTING"');
|
||
});
|
||
|
||
test('script reads and echoes routing_declined config', () => {
|
||
expect(SKILL_START_SCRIPT).toMatch(/_ROUTING_DECLINED=\$\("\$_BIN\/gstack-config" get routing_declined/);
|
||
expect(SKILL_START_SCRIPT).toContain('echo "ROUTING_DECLINED: $_ROUTING_DECLINED"');
|
||
});
|
||
|
||
test('emitted block carries the routing injection AskUserQuestion', () => {
|
||
expect(routingBlock).toContain('Add routing rules to CLAUDE.md');
|
||
expect(routingBlock).toContain("I'll invoke skills manually");
|
||
});
|
||
|
||
test('routing injection respects prior decline (gate + in-block ack)', () => {
|
||
expect(SKILL_START_SCRIPT).toContain('[ "$_ROUTING_DECLINED" = "false" ]');
|
||
expect(routingBlock).toMatch(/routing_declined.*true/);
|
||
expect(routingBlock).toContain('re-enable with `__BIN__/gstack-config set routing_declined false`');
|
||
});
|
||
|
||
test('routing injection only fires when all conditions met', () => {
|
||
// Must be: HAS_ROUTING=no AND ROUTING_DECLINED=false AND PROACTIVE_PROMPTED=yes
|
||
expect(SKILL_START_SCRIPT).toContain(
|
||
'if [ "$_HAS_ROUTING" = "no" ] && [ "$_ROUTING_DECLINED" = "false" ] && [ "$_PROACTIVE_PROMPTED" = "yes" ]; then',
|
||
);
|
||
});
|
||
|
||
test('routing section content includes key routing rules', () => {
|
||
expect(routingBlock).toContain('invoke /office-hours');
|
||
expect(routingBlock).toContain('invoke /investigate');
|
||
expect(routingBlock).toContain('invoke /ship');
|
||
expect(routingBlock).toContain('invoke /qa');
|
||
});
|
||
|
||
test('routing section uses renamed checkpoint skills (not stale /checkpoint)', () => {
|
||
expect(routingBlock).toContain('invoke /context-save');
|
||
expect(routingBlock).toContain('invoke /context-restore');
|
||
expect(routingBlock).not.toContain('invoke checkpoint');
|
||
});
|
||
|
||
test('routing section uses soft "when in doubt" policy, not hard "ALWAYS invoke"', () => {
|
||
expect(routingBlock).toContain('When in doubt, invoke the skill');
|
||
expect(routingBlock).not.toContain('Do NOT answer directly');
|
||
});
|
||
});
|
||
|
||
// --- {{DESIGN_OUTSIDE_VOICES}} resolver tests ---
|
||
|
||
describe('DESIGN_OUTSIDE_VOICES resolver', () => {
|
||
test('plan-design-review contains outside voices section', () => {
|
||
const content = readSkillUnion('plan-design-review');
|
||
expect(content).toContain('Design Outside Voices');
|
||
expect(content).toContain('CODEX_MODE: ready');
|
||
expect(content).toContain('LITMUS SCORECARD');
|
||
});
|
||
|
||
test('design-review contains outside voices section', () => {
|
||
const content = fs.readFileSync(path.join(ROOT, 'design-review', 'SKILL.md'), 'utf-8');
|
||
expect(content).toContain('Design Outside Voices');
|
||
expect(content).toContain('source audit');
|
||
});
|
||
|
||
test('design-consultation contains outside voices section', () => {
|
||
const content = readSkillUnion('design-consultation');
|
||
expect(content).toContain('Design Outside Voices');
|
||
expect(content).toContain('design direction');
|
||
});
|
||
|
||
test('consultation dispatch shares the complete brief without executing product text', () => {
|
||
const content = readSkillUnion('design-consultation');
|
||
const command = [...content.matchAll(/```bash\n([\s\S]*?)```/g)]
|
||
.map(match => match[1]).find(block => block.includes('codex exec'))!;
|
||
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'design-brief-contract-'));
|
||
const quote = (value: string) => "'" + value.replaceAll("'", "'\\''") + "'";
|
||
try {
|
||
const brief = path.join(dir, 'product.md');
|
||
const capture = path.join(dir, 'argv.json');
|
||
const marker = path.join(dir, 'must-not-execute');
|
||
const product = `A product with $(touch ${marker}), \`touch ${marker}\`, and "quotes".\nUsers: builders.\n`;
|
||
fs.writeFileSync(brief, product);
|
||
fs.writeFileSync(path.join(dir, 'codex'), `#!${process.execPath}\nrequire('fs').writeFileSync(process.env.CAPTURE, JSON.stringify(process.argv.slice(2)));\nconsole.log('Recommendation: choose a clear hierarchy because builders need to find their work.');\n`, { mode: 0o700 });
|
||
const prepare = (file: string) => command.replace("'<prepared-prompt-file>'", quote(file))
|
||
.replaceAll('$HOME/.claude/skills/gstack', ROOT);
|
||
const env = { ...process.env, PATH: dir + path.delimiter + process.env.PATH, CAPTURE: capture,
|
||
CODEX_THREAD_ID: '', CODEX_SANDBOX: '', CLAUDECODE: '1', GSTACK_ACTIVE_HOST: 'claude' };
|
||
const result = spawnSync('bash', ['-c', prepare(brief)], { cwd: ROOT, env, encoding: 'utf8', timeout: 5_000 });
|
||
expect(result.status, result.stderr).toBe(0);
|
||
const args = JSON.parse(fs.readFileSync(capture, 'utf8'));
|
||
expect(args[0]).toBe('exec');
|
||
expect(args[1]).toBe(product.trimEnd());
|
||
expect(args).toContain('read-only');
|
||
expect(fs.readFileSync(brief, 'utf8')).toBe(product);
|
||
expect(fs.existsSync(marker)).toBe(false);
|
||
fs.unlinkSync(capture);
|
||
const missing = spawnSync('bash', ['-c', prepare(path.join(dir, 'absent'))], { cwd: ROOT, env, encoding: 'utf8', timeout: 5_000 });
|
||
expect(missing.status).not.toBe(0);
|
||
expect(missing.stderr).toContain('No such file');
|
||
expect(fs.existsSync(capture)).toBe(false);
|
||
expect(content).toContain('Neither voice inherits context: give both the same brief');
|
||
expect(content).toContain('await both before synthesis');
|
||
expect(content).toContain('use only the native voice');
|
||
expect(content).toContain('give the native Agent its absolute path');
|
||
expect(content).toContain('Read the complete product brief at [the absolute DESIGN_BRIEF path printed above]');
|
||
expect(content).toContain("Check each proposed family's official Google Fonts/Fontshare listing via WebSearch/Aside for its exact name, required weights, license and loading URL");
|
||
expect(content).toContain('Omit faces you cannot verify');
|
||
expect(content).toContain('a face may serve multiple roles');
|
||
expect(content).not.toContain('a single question that covers everything');
|
||
} finally { fs.rmSync(dir, { recursive: true, force: true }); }
|
||
});
|
||
|
||
test('branches correctly per skillName — different prompts', () => {
|
||
const planContent = readSkillUnion('plan-design-review');
|
||
const consultContent = readSkillUnion('design-consultation');
|
||
// plan-design-review uses analytical prompt (high reasoning)
|
||
expect(planContent).toContain('model_reasoning_effort="high"');
|
||
// design-consultation uses creative prompt (medium reasoning)
|
||
expect(consultContent).toContain('model_reasoning_effort="medium"');
|
||
});
|
||
});
|
||
|
||
// --- {{DESIGN_HARD_RULES}} resolver tests ---
|
||
|
||
describe('DESIGN_HARD_RULES resolver', () => {
|
||
test('plan-design-review Pass 4 contains hard rules', () => {
|
||
const content = readSkillUnion('plan-design-review');
|
||
expect(content).toContain('Design Hard Rules');
|
||
expect(content).toContain('Classifier');
|
||
expect(content).toContain('MARKETING/LANDING PAGE');
|
||
expect(content).toContain('APP UI');
|
||
});
|
||
|
||
test('design-review contains hard rules', () => {
|
||
const content = fs.readFileSync(path.join(ROOT, 'design-review', 'SKILL.md'), 'utf-8');
|
||
expect(content).toContain('Design Hard Rules');
|
||
});
|
||
|
||
test('includes all 3 rule sets', () => {
|
||
const content = readSkillUnion('plan-design-review');
|
||
expect(content).toContain('Landing page rules');
|
||
expect(content).toContain('App UI rules');
|
||
expect(content).toContain('Universal rules');
|
||
});
|
||
|
||
test('classifier names the four visitor modes and keeps the legacy aliases', () => {
|
||
const content = readSkillUnion('plan-design-review');
|
||
for (const mode of ['PERSUADE', 'OPERATE', 'READ', 'EXPERIENCE', 'HYBRID']) expect(content).toContain(`**${mode}**`);
|
||
expect(content).toContain('Read rules');
|
||
expect(content).toContain('Experience rules');
|
||
expect(content).toContain('classify per section, not per page');
|
||
});
|
||
|
||
test('carries the craft-floor reflexes and the three-looks calibration', () => {
|
||
const content = readSkillUnion('plan-design-review');
|
||
expect(content).toContain('Reflexes no detector catches');
|
||
expect(content).toContain('Browser surfaces carry the design');
|
||
expect(content).toContain('One authored motion moment');
|
||
expect(content).toContain('Depth has an offset');
|
||
expect(content).toContain('Light or dark comes from the use scene');
|
||
expect(content).toContain('Calibration: the three looks');
|
||
});
|
||
|
||
test('slop section lists detector rule ids and judgment tells outside design-review', () => {
|
||
const content = readSkillUnion('plan-design-review');
|
||
expect(content).toContain('Detector rule ids for the rest of the catalog');
|
||
expect(content).toContain('nested-cards: Nested cards');
|
||
expect(content).toContain('Judgment tells with no detector rule');
|
||
// Never a bracketed gstack-only id.
|
||
expect(content).not.toContain('[hero-metrics]');
|
||
});
|
||
|
||
test('design-consultation carries the font procedure, role-scoped lists, color strategies, and catalog bullets', () => {
|
||
const content = readSkillUnion('design-consultation');
|
||
expect(content).toContain('Choosing faces: a procedure, not a menu');
|
||
expect(content).toContain('**Overused as display**');
|
||
expect(content).toContain('Fine as body/UI on an Operate or Read surface');
|
||
expect(content).toContain('**Banned in any role:** Papyrus');
|
||
expect(content).toContain('Restrained (1 accent + neutrals');
|
||
expect(content).toContain('Drenched (color as the primary design tool');
|
||
expect(content).toContain('Light vs dark is not one of the dials');
|
||
expect(content).toContain('Calibration: the three looks');
|
||
// Bullets are prose only: never a bracketed rule id in the proposal skill.
|
||
expect(content).toContain('- A card inside a card is always wrong.');
|
||
expect(content).not.toMatch(/^- \[[a-z-]+\] /m);
|
||
// The old menu is gone.
|
||
expect(content).not.toContain('Font recommendations by purpose');
|
||
});
|
||
|
||
test('design-html blacklist lines carry catalog ids', () => {
|
||
const content = fs.readFileSync(path.join(ROOT, 'design-html', 'SKILL.md'), 'utf-8');
|
||
expect(content).toContain('**Never include by default (AI slop blacklist):**');
|
||
expect(content).toContain('Purple/blue gradients as default <!-- ai-color-palette -->');
|
||
expect(content).toContain('lib/design-catalog.ts');
|
||
});
|
||
|
||
test('design-review renders the catalog once: Methodology category 9 carries it, Hard Rules points at it', () => {
|
||
const content = fs.readFileSync(path.join(ROOT, 'design-review', 'SKILL.md'), 'utf-8');
|
||
expect(content.split('### Design Hard Rules').length - 1).toBe(1);
|
||
// Category 9 lists the rule once (ids only); Typography points at the same id from its overused-face item.
|
||
expect(content.split('[overused-font]').length - 1).toBe(2);
|
||
expect(content).toContain('are Methodology category 9');
|
||
expect(content).toContain('**9. AI Slop Detection**');
|
||
expect(content).toContain('Detector rules (ids only;');
|
||
expect(content).toContain('[nested-cards] nested cards');
|
||
expect(content).toContain('Judgment tells (no detector rule');
|
||
// The legacy blacklist is not repeated as a numbered list in design-review.
|
||
expect(content).not.toMatch(/^1\. Purple\/violet\/indigo/m);
|
||
});
|
||
|
||
test('references shared AI slop blacklist items', () => {
|
||
const content = readSkillUnion('plan-design-review');
|
||
expect(content).toContain('3-column feature grid');
|
||
expect(content).toContain('Purple/violet/indigo');
|
||
});
|
||
|
||
test('includes OpenAI hard rejection criteria', () => {
|
||
const content = readSkillUnion('plan-design-review');
|
||
expect(content).toContain('Generic SaaS card grid');
|
||
expect(content).toContain('Carousel with no narrative purpose');
|
||
});
|
||
|
||
test('includes OpenAI litmus checks', () => {
|
||
const content = readSkillUnion('plan-design-review');
|
||
expect(content).toContain('Brand/product unmistakable');
|
||
expect(content).toContain('premium with all decorative shadows removed');
|
||
});
|
||
});
|
||
|
||
describe('Design approval reconciliation', () => {
|
||
test('token alignment reuses the approved outcome while preserving first decisions and new tradeoffs', () => {
|
||
const section = fs.readFileSync(path.join(ROOT, 'plan-design-review/sections/review-sections.md'), 'utf8');
|
||
const pass = extractMarkdownSection(section, '### Pass 5: Design System Alignment');
|
||
const stop = pass.indexOf('**STOP.**');
|
||
expect(stop).toBeGreaterThan(0);
|
||
const beforeQuestion = pass.slice(0, stop);
|
||
expect(beforeQuestion).toContain('check whether an earlier pass already approved that outcome');
|
||
expect(beforeQuestion).toContain('apply the established tokens and update every stale gap/reference under that decision');
|
||
expect(beforeQuestion).toContain('changing the plan location or spelling out the same fix is not a new issue');
|
||
expect(beforeQuestion).toContain('Ask again only if new evidence exposes an unresolved requirement or tradeoff, and name it');
|
||
expect(beforeQuestion).toContain('An unapproved violation still needs its first individual decision');
|
||
expect(pass.slice(stop)).toContain('AskUserQuestion once per issue. Do NOT batch. Recommend + WHY.');
|
||
});
|
||
|
||
test('loaded review section distinguishes individual issue decisions from navigation', () => {
|
||
const section = fs.readFileSync(path.join(ROOT, 'plan-design-review/sections/review-sections.md'), 'utf8');
|
||
expect(section).toContain('Complete one decision cycle per unresolved finding');
|
||
expect(section).toContain('Scope, focus, setup, and next-step choices approve no remedies.');
|
||
expect(section).toContain('Never use the final next-step AskUserQuestion to satisfy the issue-approval loop.');
|
||
expect(section).toContain('With no unresolved findings, no issue question is required.');
|
||
});
|
||
|
||
test('Design blocking Exit checklist reconciles approvals and refreshes stale output after a decision', () => {
|
||
const main = fs.readFileSync(path.join(ROOT, 'plan-design-review/SKILL.md'), 'utf8');
|
||
const check = extractMarkdownSection(main, '## Section self-check');
|
||
expect(check).toContain('Before summaries, review logs or next-step menus, run approval check 0 below.');
|
||
const gate = extractMarkdownSection(main, '## EXIT PLAN MODE GATE (BLOCKING)');
|
||
expect(gate.indexOf('0. Approvals:')).toBeGreaterThanOrEqual(0);
|
||
expect(gate.indexOf('0. Approvals:')).toBeLessThan(gate.indexOf('1. Read the plan file'));
|
||
expect(gate).toContain("each issue's remedy needs its own AskUserQuestion call and answer.");
|
||
expect(gate).toContain("Never group distinct issues.");
|
||
expect(gate).toContain('Never group distinct issues. DESIGN.md tokens and navigation are not approval.');
|
||
expect(gate).toContain('Honor prior exact decisions and preamble-authorized per-issue auto-decisions;');
|
||
expect(gate).toContain('preamble-authorized');
|
||
expect(gate).toContain('record why. Deferrals remain unresolved.');
|
||
expect(gate).toContain('If missing, reset drafts to pending, ask and wait.');
|
||
expect(gate).toContain('refresh the plan and report, pass the Read-back gate, then update the review');
|
||
expect(gate).toContain('log and rerun this gate.');
|
||
expect(gate).toContain('after your most recent write to it');
|
||
});
|
||
|
||
test('CEO readiness binds actual approvals before its read-only exit verification', () => {
|
||
const main = fs.readFileSync(path.join(ROOT, 'plan-ceo-review/SKILL.md'), 'utf8');
|
||
const check = extractMarkdownSection(main, '## Section self-check');
|
||
const gate = extractMarkdownSection(main, '## EXIT PLAN MODE GATE (BLOCKING)').replace(/\s+/g, ' ');
|
||
const section = fs.readFileSync(path.join(ROOT, 'plan-ceo-review/sections/review-sections.md'), 'utf8');
|
||
const readiness = extractMarkdownSection(section, '## Approval readiness').replace(/\s+/g, ' ');
|
||
expect(section.indexOf('## Approval readiness')).toBeLessThan(section.indexOf('## Required Outputs'));
|
||
expect(readiness).toContain('No report or completion log is needed to run this check');
|
||
expect(readiness).toContain('For each approved remedy:');
|
||
expect(readiness).toContain('its actual answer, exact prior approval or preamble-authorized per-issue auto-decision');
|
||
expect(readiness).toContain('Setup, mode and navigation are not remedy approvals');
|
||
expect(readiness).toContain('an approach approves only its explicit commitments and their directly required tests');
|
||
expect(readiness).toContain("the plan applies only that answer's scope");
|
||
expect(readiness).toContain('Independent remedies and additional verification choices need their own rows and answers');
|
||
expect(readiness).toContain('Keep declined, deferred and unanswered changes out of accepted work');
|
||
expect(readiness).toContain('An approved delivery-scope deferral is settled');
|
||
expect(readiness).toContain('Deferring a needed policy or remedy decision leaves that choice unresolved; show it in the final report');
|
||
expect(readiness).toContain('If a draft lacks approval, mark it pending and use 0D; repeat this check after its answer');
|
||
expect(readiness).toContain('At the end of the six-column decision ledger, record `Approval readiness: PASS`');
|
||
expect(readiness).toContain('the checked row IDs and their actual answer or approval references');
|
||
expect(readiness).toContain('Save or present the updated plan under Step 0');
|
||
expect(readiness).toContain('A substantive change invalidates this result; navigation alone does not');
|
||
const normalizedMain = main.replace(/\s+/g, ' ');
|
||
expect(normalizedMain).toContain('Default to implementation-ready');
|
||
expect(normalizedMain).toContain('Use strategy-only only when the user asks for');
|
||
expect(normalizedMain).toContain('Use one narrow decision only when the user names a single choice');
|
||
expect(gate).toContain('Verify `Approval readiness: PASS` against current row IDs and answer references');
|
||
expect(gate).toContain('Read-only verification');
|
||
expect(gate.indexOf('Verify `Approval readiness: PASS`')).toBeLessThan(gate.indexOf('1. Read the plan file'));
|
||
expect(gate).toContain('If stale because a choice changed');
|
||
expect(gate.replace(/\s+/g, ' ')).toContain('return to 0D for that choice only; then repeat readiness, affected outputs, report Read-back, Review Log and dashboard');
|
||
expect(gate).toContain('Failed checks use **Gate outcome: Blocked**');
|
||
expect(main).toContain('**Pass with log-only gaps:**');
|
||
expect(gate).toContain('end without success telemetry, ExitPlanMode or the queued handoff');
|
||
expect(check.replace(/\s+/g, ' ')).toContain('Confirm you Read `sections/review-sections.md` and executed Sections 1–10');
|
||
expect(check.replace(/\s+/g, ' ')).toContain("Section 11's findings or no-UI skip, required outputs and report from that file");
|
||
expect(check).toContain('stop, Read and redo the review');
|
||
const decisions = main.slice(main.indexOf('### 0D.'), main.indexOf('### 0E.')).replace(/\s+/g, ' ');
|
||
expect(decisions).toMatch(/Ask one row per call with that object unchanged, without recomposing/);
|
||
expect(decisions).toContain("**Choose the question's route first:**");
|
||
expect(decisions).toContain('**Admin question:** mode, setup, navigation, document approval or promotion');
|
||
const admin = decisions.split('- **Admin question:**')[1]!.split('- **Plan decision:**')[0]!;
|
||
expect(admin).toContain('Use its listed menu and the preamble question transport, then wait and record the answer');
|
||
expect(admin).toContain('Skip steps 1–4; this approves no plan changes');
|
||
expect(decisions).toContain('**Plan decision:** review-depth expansion, scope additions/cuts, approach choices, TODOs, specs and review/outside findings');
|
||
expect(decisions).toContain('Start at step 1. Reuse exact prior approvals; run steps 2–4 only when a new answer is needed, even for one option');
|
||
expect(decisions).toContain('If an admin answer requests a plan change, use the Plan decision route for that change');
|
||
expect(decisions).toContain('Compare its question, header, labels and full descriptions literally with the verified fields');
|
||
expect(decisions).toContain('`D<N> — <ROW-ID>: <one-line question>`');
|
||
const row = decisions.slice(decisions.indexOf('**Pre-question checkpoint:**'), decisions.indexOf('Effort/risk must'));
|
||
expect(row).toMatch(/Find exactly one (?:ledger )?row/);
|
||
for (const field of ['ID', 'step 2', 'owner', 'Current/Proposed', 'Status', 'Exact approval and scope']) expect(row).toContain(field);
|
||
expect(decisions).toContain('**STOP for the actual answer, even for a lone option.**');
|
||
expect(decisions).toMatch(/amend only (?:what it authorizes|authorized work)/);
|
||
expect(decisions).toContain('Reuse exact approvals. Reopen only for contradictions, changed assumptions or user instructions, never speculation or reviewer agreement');
|
||
expect(decisions).toMatch(/reopen only|requires reopening it/i);
|
||
expect(decisions).toContain('Record findings even after resolution');
|
||
expect(decisions).toContain('say "No issues, moving on." only with none');
|
||
const outside = extractMarkdownSection(section, '**Cross-model tension:**').replace(/\s+/g, ' ');
|
||
expect(outside).toContain('compare reviews only if an external reviewer completed');
|
||
expect(outside).toContain('For a same-harness/native fallback, skip this comparison');
|
||
expect(outside).toContain('do not write a CROSS-MODEL line');
|
||
const findings = extractMarkdownSection(section, '**Integrate reviewer findings:**').replace(/\s+/g, ' ');
|
||
expect(findings).toContain('either an external reviewer or the bounded native fallback completed with a valid report');
|
||
expect(findings).toContain('Native fallback findings count as findings from the current harness');
|
||
expect(findings).toContain('Apply Outside Voice Integration Rule to every finding');
|
||
});
|
||
|
||
test('Eng cannot exit with unasked findings listed only in an unresolved-decisions report', () => {
|
||
const main = fs.readFileSync(path.join(ROOT, 'plan-eng-review/SKILL.md'), 'utf8');
|
||
const check = extractMarkdownSection(main, '## Section self-check');
|
||
const rawGate = extractMarkdownSection(main, '## EXIT PLAN MODE GATE (BLOCKING)');
|
||
const gate = rawGate.replace(/\s+/g, ' ');
|
||
const section = fs.readFileSync(path.join(ROOT, 'plan-eng-review/sections/review-sections.md'), 'utf8');
|
||
const readiness = extractMarkdownSection(section, '## Approval readiness');
|
||
const normalizedReadiness = readiness.replace(/\s+/g, ' ');
|
||
expect(section.indexOf('## Approval readiness')).toBeLessThan(section.indexOf('## Required outputs'));
|
||
expect(normalizedReadiness).toContain('Only the ledger is needed here; completion outputs and logs come next');
|
||
expect(normalizedReadiness).toContain('At the end of `## Decision ledger`, record `Approval readiness: PASS`');
|
||
expect(normalizedReadiness).toContain('checked IDs and actual answer references');
|
||
expect(normalizedReadiness).toContain('A substantive change invalidates this result; navigation alone does not');
|
||
expect(readiness).not.toMatch(/^ {3}\S/m);
|
||
expect(gate).toContain('Confirm Approval readiness passed for the current decisions');
|
||
expect(gate).toContain('read-only verification, not a new approval or output-writing step');
|
||
const approvalParagraph = rawGate.slice(rawGate.indexOf('Confirm Approval readiness'), rawGate.indexOf('Verify all five checks'));
|
||
expect(approvalParagraph).not.toMatch(/^ {3}\S/m);
|
||
expect([...rawGate.matchAll(/^([1-5])\. /gm)].map(match => match[1])).toEqual(['1', '2', '3', '4', '5']);
|
||
expect(gate.indexOf('Confirm Approval readiness')).toBeLessThan(gate.indexOf('1. Read the report file'));
|
||
expect(normalizedReadiness).toContain('check the ledger against every accepted remedy');
|
||
expect(normalizedReadiness).toContain('Each must cite its own actual answer, exact prior approval or authorized auto-decision');
|
||
expect(normalizedReadiness).toContain('setup, mode, approach and navigation do not count');
|
||
expect(normalizedReadiness).toContain('Deferrals remain unresolved');
|
||
expect(normalizedReadiness).toContain('Carry forward an exact approved regression contract');
|
||
expect(normalizedReadiness).toContain('Otherwise, its behavior and assertions need one dedicated decision');
|
||
expect(readiness).not.toContain('REGRESSION test is already authorized');
|
||
expect(normalizedReadiness).toContain('If approval is missing, mark that draft pending, resolve the choice through Decision procedure and repeat this check');
|
||
expect(normalizedReadiness).toContain('Continue to Required outputs, preserving unresolved decisions in the report');
|
||
// Readiness verifies evidence; the one procedure owns asking and applying.
|
||
// The procedure contains a fenced H2 ledger example; use its real next step.
|
||
const decisionStart = section.indexOf('\n## Decision procedure\n');
|
||
const decisionEnd = section.indexOf('\n## Scope Challenge\n', decisionStart);
|
||
expect(decisionStart).toBeGreaterThan(0);
|
||
expect(decisionEnd).toBeGreaterThan(decisionStart);
|
||
const decisions = section.slice(decisionStart, decisionEnd).replace(/\s+/g, ' ');
|
||
expect(decisions).toContain('AskUserQuestion({ questions: [currentDecision] })');
|
||
expect(decisions).toContain('one question object for one choice; other IDs wait');
|
||
expect(decisions).toContain("If you discover another independent choice, separate it and rebuild this comparison before saving or sending the question");
|
||
expect(decisions.indexOf("**STOP until the actual answer arrives.**")).toBeLessThan(decisions.indexOf('### Record the answer'));
|
||
expect(decisions).toContain("For the next choice, use the updated working plan and answer");
|
||
expect(gate).toContain('report the stale verification and stop');
|
||
expect(gate).toContain('Resume under **Recovery routing → Late change or missing work**');
|
||
const recovery = readSkillUnion('plan-eng-review').split('**Late change or missing work:**')[1]!.split('**Blocked outcome:**')[0]!.replace(/\s+/g, ' ');
|
||
expect(recovery).toContain('new or reopened choices use Decision procedure');
|
||
expect(recovery).toContain('Repeat Approval readiness, then Required outputs steps 1–4 for changed outputs before choosing navigation again');
|
||
expect(gate).toContain('all six columns: Review / Trigger / Why / Runs / Status / Findings');
|
||
expect(gate).toContain('follow **Blocked outcome**');
|
||
const report = extractMarkdownSection(section, '### Write to the report file');
|
||
expect(report).toContain('Then follow **Blocked outcome** in the entrypoint.');
|
||
expect(report).toContain('report the error and follow **Blocked outcome** before Review Log or decision logging');
|
||
});
|
||
|
||
test('approval entry does not alter other review Exit checklists', () => {
|
||
for (const skill of ['plan-devex-review']) {
|
||
const gate = extractMarkdownSection(readSkillUnion(skill), '## EXIT PLAN MODE GATE (BLOCKING)');
|
||
expect(gate).not.toContain('0. Approvals:');
|
||
expect(gate).toContain('1. Read the plan file');
|
||
expect(gate).toContain('5. If a plan file is in context');
|
||
}
|
||
});
|
||
|
||
test('task and decision reporting require approval while preserving unanswered findings', () => {
|
||
const section = fs.readFileSync(path.join(ROOT, 'plan-design-review/sections/review-sections.md'), 'utf8');
|
||
const reconcile = section.indexOf('Before synthesizing tasks or the completion summary');
|
||
expect(reconcile).toBeGreaterThan(0);
|
||
expect(reconcile).toBeLessThan(section.indexOf('## Implementation Tasks'));
|
||
expect(section).toContain('Export only agreed implementation work; retain unapproved remedies as pending findings.');
|
||
expect(section).toContain('Count only individually approved new decisions');
|
||
expect(section).toContain('including a finding not yet asked');
|
||
});
|
||
});
|
||
|
||
// --- {{DESIGN_DETECTOR}} resolver tests ---
|
||
|
||
describe('DESIGN_DETECTOR resolver', () => {
|
||
const designReview = () => fs.readFileSync(path.join(ROOT, 'design-review', 'SKILL.md'), 'utf-8');
|
||
const designHtml = () => fs.readFileSync(path.join(ROOT, 'design-html', 'SKILL.md'), 'utf-8');
|
||
const bashBlocksOf = (content: string) => [...content.matchAll(/```bash\n([\s\S]*?)```/g)].map(m => m[1]);
|
||
|
||
test('design-review carries the probe, Phase 0, the DOM dump, and the run id', () => {
|
||
const c = designReview();
|
||
expect(c).toContain('gstack-design-detect.ts probe --host claude');
|
||
expect(c).toContain('IMPECCABLE_READY');
|
||
// the consent-gated install: offered once, only on the probe's say-so, never in spawned sessions, never via npx
|
||
expect(c).toContain('DESIGN_DETECTOR_INSTALL_OFFER');
|
||
expect(c).toContain('gstack-design-detect.ts install --host claude');
|
||
expect(c).toContain("Install impeccable's design detector engine?");
|
||
expect(c).toContain('gstack-config set design_detector_install_prompted true');
|
||
expect(c).toContain('`SESSION_KIND: spawned` or a headless run, never install and never ask');
|
||
expect(c).toContain('**Phase 0: mechanical scan**');
|
||
expect(c).toContain('scan --changed <base> --format gstack --host claude');
|
||
expect(c).toContain('### DOM dump (DOM mode only');
|
||
expect(c).toContain('data-gstack-dom-css');
|
||
expect(c).toContain(`$B js '('"$_DUMP"')()' --out "$_TMP/{page}.dom.html" --raw`);
|
||
expect(c).toContain('DOM_DUMP_OK');
|
||
expect(c).toContain('DOM_DUMP_REDACTION_BLOCKED');
|
||
expect(c).toContain('DOM_DUMP_TOO_LARGE');
|
||
expect(c).toContain('REPORT_DIR="$GSTACK_STATE_ROOT/projects/$SLUG/designs/design-audit-$(date +%Y%m%d)"');
|
||
expect(c).toContain('RUN_ID="$(date +%H%M%S)-$$"');
|
||
expect(c).toContain('"schemaVersion": 2');
|
||
expect(c).toContain('engine changed X → Y; rule set may differ');
|
||
expect(c).toContain('Detector: N → M');
|
||
expect(c).toContain('/impeccable typeset');
|
||
});
|
||
|
||
test('the DOM-dump script is loaded from lib/dom-dump.js, never inlined in the prose', () => {
|
||
const c = designReview();
|
||
expect(c).not.toMatch(/```js\n/);
|
||
expect(c).not.toContain('document.documentElement.cloneNode');
|
||
expect(c).toContain('_DUMP=$(cat "$HOME/.claude/skills/gstack/lib/dom-dump.js")');
|
||
expect(c).toContain(`const html = await pg.evaluate('"$_DUMP"');`);
|
||
expect(c).toContain('_TMP=$(mktemp -d); _DUMP=$(cat "$HOME/.claude/skills/gstack/lib/dom-dump.js")');
|
||
});
|
||
|
||
test('every rendered Aside script is single-quoted: a page-controlled <url> is never inside a double-quoted bash string', () => {
|
||
const files = [...fs.readdirSync(ROOT).filter(d => fs.existsSync(path.join(ROOT, d, 'SKILL.md'))).map(d => path.join(ROOT, d, 'SKILL.md')),
|
||
...fs.readdirSync(ROOT).flatMap(d => fs.existsSync(path.join(ROOT, d, 'sections')) ? fs.readdirSync(path.join(ROOT, d, 'sections')).filter(f => f.endsWith('.md')).map(f => path.join(ROOT, d, 'sections', f)) : [])];
|
||
expect(files.length).toBeGreaterThan(10);
|
||
for (const f of files) {
|
||
const c = fs.readFileSync(f, 'utf-8');
|
||
expect(c, path.relative(ROOT, f)).not.toMatch(/^aside repl "/m);
|
||
}
|
||
});
|
||
|
||
test('the E2E fixture slice markers exist in the rendered design skills (a template rename fails here, not in paid CI)', () => {
|
||
const dr = designReview();
|
||
const dh = fs.readFileSync(path.join(ROOT, 'design-html', 'SKILL.md'), 'utf-8');
|
||
for (const [a, b] of [['**Design detector (optional, deterministic):**', '**Create output directories:**'], ['**Phase 0: mechanical scan**', '## Phases 1-6'], ['### DOM dump (DOM mode only', '### Auth Detection']]) {
|
||
expect(sliceBetween(dr, a, b).length, `${a} .. ${b}`).toBeGreaterThan(100);
|
||
}
|
||
for (const [a, b] of [['**Design detector (optional, deterministic):**', '## Step 0: Input Detection'], ['### Slop Gate (bounded, never a loop)', '### Verification Screenshots']]) {
|
||
expect(sliceBetween(dh, a, b).length, `${a} .. ${b}`).toBeGreaterThan(100);
|
||
}
|
||
});
|
||
|
||
test('design-html carries the probe and the bounded slop gate', () => {
|
||
const c = designHtml();
|
||
expect(c).toContain('gstack-design-detect.ts probe --host claude');
|
||
expect(c).toContain('### Slop Gate (bounded, never a loop)');
|
||
expect(c).toContain('One pass, not a loop.');
|
||
expect(c).toContain('impeccable-disable <rule>: <reason>');
|
||
});
|
||
|
||
test('ship and review unions reach the detector through review-lite and the checklist', () => {
|
||
const ship = readSkillUnion('ship');
|
||
expect(ship).toContain('**Mechanical pass first.**');
|
||
expect(ship).toContain('scan --changed <base> --format gstack --host claude');
|
||
expect(ship).toContain('"detector":D');
|
||
expect(ship).toContain('Detector: "clean" | "N findings');
|
||
const review = readSkillUnion('review');
|
||
expect(review).toContain('run the mechanical pass at the top of that checklist');
|
||
const checklist = fs.readFileSync(path.join(ROOT, 'review', 'design-checklist.md'), 'utf-8');
|
||
expect(checklist).toContain('**0. Mechanical pass first.**');
|
||
expect(checklist).toContain('IMPECCABLE_READY');
|
||
});
|
||
|
||
test('every rendered invocation uses bun --no-env-file and ends a scan with the exit echo; no bash block runs npx impeccable', () => {
|
||
for (const content of [designReview(), designHtml(), readSkillUnion('ship'), readSkillUnion('review'), fs.readFileSync(path.join(ROOT, 'review', 'design-checklist.md'), 'utf-8')]) {
|
||
for (const block of bashBlocksOf(content)) {
|
||
expect(block).not.toContain('npx impeccable');
|
||
for (const line of block.split('\n')) {
|
||
if (!line.includes('gstack-design-detect.ts')) continue;
|
||
expect(line).toContain('bun --no-env-file run ');
|
||
if (/gstack-design-detect\.ts scan /.test(line)) expect(line).toContain('echo "DETECT_EXIT_CODE=$?"');
|
||
}
|
||
}
|
||
}
|
||
});
|
||
|
||
test('--host is rendered per host', () => {
|
||
// Fresh codex render into a temp out-dir: the tracked tree is Claude-only and
|
||
// the gitignored .agents/ copy may be stale.
|
||
const out = fs.mkdtempSync(path.join(os.tmpdir(), 'gstack-detector-host-'));
|
||
try {
|
||
const r = Bun.spawnSync(['bun', 'run', 'scripts/gen-skill-docs.ts', '--host', 'codex', '--out-dir', out], { cwd: ROOT, timeout: 120_000 });
|
||
expect(r.exitCode).toBe(0);
|
||
const codex = fs.readFileSync(path.join(out, '.agents', 'skills', 'gstack-design-review', 'SKILL.md'), 'utf-8');
|
||
expect(codex).toContain('gstack-design-detect.ts probe --host codex');
|
||
expect(codex).not.toContain('probe --host claude');
|
||
expect(codex).toContain('$GSTACK_ROOT/lib/dom-dump.js');
|
||
} finally {
|
||
fs.rmSync(out, { recursive: true, force: true });
|
||
}
|
||
});
|
||
});
|
||
|
||
// --- {{DESIGN_MD_CHECK}} resolver + open DESIGN.md adoption ---
|
||
|
||
describe('DESIGN_MD_CHECK resolver and open DESIGN.md adoption', () => {
|
||
test('design-consultation asks the conversion question once and writes the spec form', () => {
|
||
const c = readSkillUnion('design-consultation');
|
||
expect(c).toContain('gstack-design-md.ts check DESIGN.md');
|
||
expect(c).toContain('DESIGN_MD_FORMAT: spec');
|
||
expect(c).toContain('mark legacy-keep');
|
||
expect(c).toContain('convert --write');
|
||
expect(c).toContain('# gstack: design-md-format=spec');
|
||
expect(c).toContain("## Do's and Don'ts");
|
||
expect(c).toContain('## Elevation & Depth');
|
||
expect(c).toContain('fontFeature: tnum');
|
||
expect(c).toContain('"{colors.primary}"');
|
||
// the legacy template is gone
|
||
expect(c).not.toContain('## Product Context\n- **What this is:**');
|
||
});
|
||
|
||
test('design-review calibrates against tokens and never re-offers conversion; design-html writes the spec form', () => {
|
||
const dr = fs.readFileSync(path.join(ROOT, 'design-review', 'SKILL.md'), 'utf-8');
|
||
expect(dr).toContain('gstack-design-md.ts check DESIGN.md');
|
||
expect(dr).toContain('gstack-design-md.ts tokens DESIGN.md');
|
||
expect(dr).toContain('never offer a conversion here');
|
||
expect(dr).not.toContain('mark legacy-keep');
|
||
const dh = fs.readFileSync(path.join(ROOT, 'design-html', 'SKILL.md'), 'utf-8');
|
||
expect(dh).toContain('# gstack: design-md-format=spec');
|
||
const pdr = readSkillUnion('plan-design-review');
|
||
expect(pdr).toContain('{colors.primary}');
|
||
const checklist = fs.readFileSync(path.join(ROOT, 'review', 'design-checklist.md'), 'utf-8');
|
||
expect(checklist).toContain('gstack-design-md.ts tokens DESIGN.md');
|
||
expect(readSkillUnion('ship')).toContain('gstack-design-md.ts tokens DESIGN.md');
|
||
});
|
||
|
||
test('every rendered gstack-design-md invocation uses bun --no-env-file', () => {
|
||
for (const content of [readSkillUnion('design-consultation'), fs.readFileSync(path.join(ROOT, 'design-review', 'SKILL.md'), 'utf-8'), readSkillUnion('ship')]) {
|
||
for (const line of content.split('\n')) {
|
||
if (line.includes('gstack-design-md.ts')) expect(line).toContain('bun --no-env-file run ');
|
||
}
|
||
}
|
||
});
|
||
});
|
||
|
||
// --- PRODUCT.md prefill + /impeccable handoffs ---
|
||
|
||
describe('PRODUCT.md prefill and /impeccable handoffs', () => {
|
||
test('design-consultation and design-shotgun read PRODUCT.md and never open the impeccable skill', () => {
|
||
for (const skill of ['design-consultation', 'design-shotgun']) {
|
||
const c = readSkillUnion(skill);
|
||
expect(c).toContain('cat PRODUCT.md 2>/dev/null | head -120 || echo "NO_PRODUCT_MD"');
|
||
expect(c).toContain('do not re-ask');
|
||
expect(c).toContain('Never open `.claude/skills/impeccable/**`');
|
||
}
|
||
});
|
||
|
||
test('handoffs are gated on IMPECCABLE_SKILL: present in review-lite and design-review', () => {
|
||
expect(readSkillUnion('ship')).toContain('IMPECCABLE_SKILL: present`, end each NEEDS INPUT detector row with the `handoff=` command');
|
||
const dr = fs.readFileSync(path.join(ROOT, 'design-review', 'SKILL.md'), 'utf-8');
|
||
expect(dr).toContain('a deferred one ends with its `handoff=` command when `IMPECCABLE_SKILL: present`');
|
||
expect(dr).toContain('skip every detector step, including `/impeccable` handoff lines');
|
||
});
|
||
});
|
||
|
||
// --- Extended DESIGN_SKETCH resolver tests ---
|
||
|
||
describe('DESIGN_SKETCH extended with outside voices', () => {
|
||
const content = readSkillUnion('office-hours'); // carved: Phase 5/6 prose moved to section
|
||
|
||
test('contains outside design voices step', () => {
|
||
expect(content).toContain('Outside design voices');
|
||
});
|
||
|
||
test('offers opt-in via AskUserQuestion', () => {
|
||
expect(content).toContain('outside design perspectives');
|
||
});
|
||
|
||
test('still contains original wireframe steps', () => {
|
||
expect(content).toContain('wireframe');
|
||
expect(content).toContain('gstack-render.ts');
|
||
});
|
||
});
|
||
|
||
// --- Extended DESIGN_REVIEW_LITE resolver tests ---
|
||
|
||
describe('DESIGN_REVIEW_LITE extended with Codex', () => {
|
||
const content = readShipUnion();
|
||
|
||
test('contains Codex design voice block', () => {
|
||
expect(content).toContain('Codex design voice');
|
||
expect(content).toContain('CODEX (design)');
|
||
});
|
||
|
||
test('still contains original checklist steps', () => {
|
||
expect(content).toContain('design-checklist.md');
|
||
expect(content).toContain('SCOPE_FRONTEND');
|
||
});
|
||
|
||
test('design-checklist path uses installed gstack/review root (#2694)', () => {
|
||
// #2694: generateDesignReviewLite used to emit
|
||
// `.claude/skills/review/design-checklist.md` (missing the gstack/ segment).
|
||
// After install the file lives at ~/.claude/skills/gstack/review/design-checklist.md.
|
||
// The bad relative form must not appear — the good path does not contain it
|
||
// as a substring because `gstack/` sits between `skills/` and `review/`.
|
||
expect(content).toContain('~/.claude/skills/gstack/review/design-checklist.md');
|
||
expect(content).not.toContain('.claude/skills/review/design-checklist.md');
|
||
});
|
||
|
||
});
|
||
|
||
// ─── Codex Generation Tests ─────────────────────────────────
|
||
|
||
describe('Codex generation (--host codex)', () => {
|
||
// .agents/ is gitignored (v0.11.2.0) — read the module-level out-dir render
|
||
// (--host all covers codex) instead of regenerating the live tree in place.
|
||
const AGENTS_DIR = path.join(EXTERNAL_OUT, '.agents', 'skills');
|
||
|
||
// Dynamic discovery of expected Codex skills: all templates except /codex.
|
||
// The out-dir is a fresh mkdtemp, so the vendored-dev-mode symlink loop
|
||
// (.agents/skills/{name} → repo root) that made the generator skip skills
|
||
// in-place can never occur here — every template renders.
|
||
const CODEX_SKILLS = (() => {
|
||
const skills: Array<{ dir: string; codexName: string }> = [];
|
||
if (fs.existsSync(path.join(ROOT, 'SKILL.md.tmpl'))) {
|
||
skills.push({ dir: '.', codexName: 'gstack' });
|
||
}
|
||
for (const entry of fs.readdirSync(ROOT, { withFileTypes: true })) {
|
||
if (!entry.isDirectory() || entry.name.startsWith('.') || entry.name === 'node_modules') continue;
|
||
if (entry.name === 'codex') continue; // /codex is excluded from Codex output
|
||
if (!fs.existsSync(path.join(ROOT, entry.name, 'SKILL.md.tmpl'))) continue;
|
||
const codexName = entry.name.startsWith('gstack-') ? entry.name : `gstack-${entry.name}`;
|
||
skills.push({ dir: entry.name, codexName });
|
||
}
|
||
return skills;
|
||
})();
|
||
|
||
test('--host codex generates correct output paths', () => {
|
||
for (const skill of CODEX_SKILLS) {
|
||
const skillMd = path.join(AGENTS_DIR, skill.codexName, 'SKILL.md');
|
||
expect(fs.existsSync(skillMd)).toBe(true);
|
||
}
|
||
});
|
||
|
||
test('root gstack bundle has OpenAI metadata for Codex skill browsing', () => {
|
||
const rootMetadata = path.join(ROOT, 'agents', 'openai.yaml');
|
||
expect(fs.existsSync(rootMetadata)).toBe(true);
|
||
const content = fs.readFileSync(rootMetadata, 'utf-8');
|
||
expect(content).toContain('display_name: "gstack"');
|
||
expect(content).toContain('Use $gstack to locate the bundled gstack skills.');
|
||
expect(content).toContain('allow_implicit_invocation: true');
|
||
});
|
||
|
||
test('externalSkillName mapping: root is gstack, others are gstack-{dir}', () => {
|
||
// Root → gstack
|
||
expect(fs.existsSync(path.join(AGENTS_DIR, 'gstack', 'SKILL.md'))).toBe(true);
|
||
// Subdirectories → gstack-{dir}
|
||
expect(fs.existsSync(path.join(AGENTS_DIR, 'gstack-review', 'SKILL.md'))).toBe(true);
|
||
expect(fs.existsSync(path.join(AGENTS_DIR, 'gstack-ship', 'SKILL.md'))).toBe(true);
|
||
// gstack-upgrade doesn't double-prefix
|
||
expect(fs.existsSync(path.join(AGENTS_DIR, 'gstack-upgrade', 'SKILL.md'))).toBe(true);
|
||
// No double-prefix: gstack-gstack-upgrade must NOT exist
|
||
expect(fs.existsSync(path.join(AGENTS_DIR, 'gstack-gstack-upgrade', 'SKILL.md'))).toBe(false);
|
||
});
|
||
|
||
test('Codex frontmatter has ONLY name + description', () => {
|
||
for (const skill of CODEX_SKILLS) {
|
||
const content = fs.readFileSync(path.join(AGENTS_DIR, skill.codexName, 'SKILL.md'), 'utf-8');
|
||
expect(content.startsWith('---\n')).toBe(true);
|
||
const fmEnd = content.indexOf('\n---', 4);
|
||
expect(fmEnd).toBeGreaterThan(0);
|
||
const frontmatter = content.slice(4, fmEnd);
|
||
// Must have name and description
|
||
expect(frontmatter).toContain('name:');
|
||
expect(frontmatter).toContain('description:');
|
||
// Must NOT have allowed-tools, version, or hooks
|
||
expect(frontmatter).not.toContain('allowed-tools:');
|
||
expect(frontmatter).not.toContain('version:');
|
||
expect(frontmatter).not.toContain('hooks:');
|
||
}
|
||
});
|
||
|
||
test('all Codex skills have agents/openai.yaml metadata', () => {
|
||
for (const skill of CODEX_SKILLS) {
|
||
const metadata = path.join(AGENTS_DIR, skill.codexName, 'agents', 'openai.yaml');
|
||
expect(fs.existsSync(metadata)).toBe(true);
|
||
const content = fs.readFileSync(metadata, 'utf-8');
|
||
expect(content).toContain(`display_name: "${skill.codexName}"`);
|
||
expect(content).toContain('short_description:');
|
||
expect(content).toContain('allow_implicit_invocation: true');
|
||
}
|
||
});
|
||
|
||
test('no .claude/skills/ in Codex output', () => {
|
||
for (const skill of CODEX_SKILLS) {
|
||
const content = fs.readFileSync(path.join(AGENTS_DIR, skill.codexName, 'SKILL.md'), 'utf-8');
|
||
expect(content).not.toContain('.claude/skills');
|
||
}
|
||
});
|
||
|
||
test('no ~/.claude/ runtime paths in Codex output', () => {
|
||
for (const skill of CODEX_SKILLS) {
|
||
const content = fs.readFileSync(path.join(AGENTS_DIR, skill.codexName, 'SKILL.md'), 'utf-8');
|
||
// Outside prompts may explicitly forbid reading Claude's skill directory.
|
||
// Every executable/runtime path must still use the selected host root.
|
||
const withoutBoundary = content.replace(/^.*IMPORTANT: Do NOT read or execute[^\n]*$/gm, '');
|
||
expect(withoutBoundary).not.toContain('~/.claude/');
|
||
}
|
||
});
|
||
|
||
test('/codex skill excluded from Codex output', () => {
|
||
expect(fs.existsSync(path.join(AGENTS_DIR, 'gstack-codex', 'SKILL.md'))).toBe(false);
|
||
expect(fs.existsSync(path.join(AGENTS_DIR, 'gstack-codex'))).toBe(false);
|
||
});
|
||
|
||
test('Codex output includes the renamed Claude Code skill and restricted runner', () => {
|
||
const content = readExternalSkillUnion(EXTERNAL_OUT, '.agents', 'claude-code');
|
||
expect(content).toContain('name: claude-code');
|
||
expect(content).toContain('gstack-claude-code');
|
||
expect(content).toContain('--access none');
|
||
expect(content).toContain('--access read-only');
|
||
expect(content).toContain('--timeout-ms');
|
||
expect(content).toContain('claude-session-id');
|
||
expect(content).not.toContain('$HOME/.claude/.credentials.json');
|
||
expect(fs.existsSync(path.join(AGENTS_DIR, 'gstack-claude'))).toBe(false);
|
||
});
|
||
|
||
test('Codex ship and review preserve outside review through Claude Code', () => {
|
||
for (const skill of ['ship', 'review']) {
|
||
const content = readExternalSkillUnion(EXTERNAL_OUT, '.agents', skill);
|
||
expect(content).not.toMatch(/codex\s+(?:exec|review)\s/);
|
||
expect(content).toContain('gstack-claude-code');
|
||
expect(content).toContain('codex_reviews');
|
||
expect(content).toContain('adversarial-review');
|
||
}
|
||
});
|
||
|
||
test('--host codex --dry-run freshness', () => {
|
||
// Dry-run against the out-dir render: determinism/idempotency check
|
||
// (regenerating produces the same bytes the module-level render did).
|
||
const result = Bun.spawnSync(['bun', 'run', 'scripts/gen-skill-docs.ts', '--host', 'codex', '--dry-run', '--out-dir', EXTERNAL_OUT], {
|
||
cwd: ROOT,
|
||
stdout: 'pipe',
|
||
stderr: 'pipe',
|
||
timeout: 120_000,
|
||
});
|
||
expect(result.exitCode).toBe(0);
|
||
const output = result.stdout.toString();
|
||
// Every Codex skill should be FRESH
|
||
for (const skill of CODEX_SKILLS) {
|
||
expect(output).toContain(`FRESH: .agents/skills/${skill.codexName}/SKILL.md`);
|
||
}
|
||
expect(output).not.toContain('STALE');
|
||
});
|
||
|
||
test('--host agents alias produces same output as --host codex', () => {
|
||
const codexResult = Bun.spawnSync(['bun', 'run', 'scripts/gen-skill-docs.ts', '--host', 'codex', '--dry-run', '--out-dir', EXTERNAL_OUT], {
|
||
cwd: ROOT,
|
||
stdout: 'pipe',
|
||
stderr: 'pipe',
|
||
timeout: 120_000,
|
||
});
|
||
const agentsResult = Bun.spawnSync(['bun', 'run', 'scripts/gen-skill-docs.ts', '--host', 'agents', '--dry-run', '--out-dir', EXTERNAL_OUT], {
|
||
cwd: ROOT,
|
||
stdout: 'pipe',
|
||
stderr: 'pipe',
|
||
timeout: 120_000,
|
||
});
|
||
expect(codexResult.exitCode).toBe(0);
|
||
expect(agentsResult.exitCode).toBe(0);
|
||
// Both should produce the same output (same FRESH lines)
|
||
expect(codexResult.stdout.toString()).toBe(agentsResult.stdout.toString());
|
||
});
|
||
|
||
test('multiline descriptions preserved in Codex output', () => {
|
||
// office-hours has a multiline description — verify it survives the frontmatter transform
|
||
const content = fs.readFileSync(path.join(AGENTS_DIR, 'gstack-office-hours', 'SKILL.md'), 'utf-8');
|
||
const fmEnd = content.indexOf('\n---', 4);
|
||
const frontmatter = content.slice(4, fmEnd);
|
||
// Description should span multiple lines (block scalar)
|
||
const descLines = frontmatter.split('\n').filter(l => l.startsWith(' '));
|
||
expect(descLines.length).toBeGreaterThan(1);
|
||
// Verify key phrases survived
|
||
expect(frontmatter).toContain('YC Office Hours');
|
||
});
|
||
|
||
test('hook skills have safety prose and no hooks: in frontmatter', () => {
|
||
const HOOK_SKILLS = ['gstack-careful', 'gstack-freeze', 'gstack-guard'];
|
||
for (const skillName of HOOK_SKILLS) {
|
||
const content = fs.readFileSync(path.join(AGENTS_DIR, skillName, 'SKILL.md'), 'utf-8');
|
||
// Must have safety advisory prose
|
||
expect(content).toContain('Safety Advisory');
|
||
// Must NOT have hooks: in frontmatter
|
||
const fmEnd = content.indexOf('\n---', 4);
|
||
const frontmatter = content.slice(4, fmEnd);
|
||
expect(frontmatter).not.toContain('hooks:');
|
||
}
|
||
});
|
||
|
||
test('all Codex SKILL.md files have auto-generated header', () => {
|
||
for (const skill of CODEX_SKILLS) {
|
||
const content = fs.readFileSync(path.join(AGENTS_DIR, skill.codexName, 'SKILL.md'), 'utf-8');
|
||
expect(content).toContain('AUTO-GENERATED from SKILL.md.tmpl');
|
||
expect(content).toContain('Regenerate: bun run gen:skill-docs');
|
||
}
|
||
});
|
||
|
||
test('Codex preamble resolves runtime assets from repo-local or global gstack roots', () => {
|
||
// Check a skill that has a preamble (review is a good candidate)
|
||
const content = fs.readFileSync(path.join(AGENTS_DIR, 'gstack-review', 'SKILL.md'), 'utf-8');
|
||
expect(content).toContain('GSTACK_ROOT');
|
||
expect(content).toContain('$_ROOT/.agents/skills/gstack');
|
||
// Phase 1/2: config reads moved into gstack-skill-start — the fence itself
|
||
// is the bin asset the preamble must resolve through $GSTACK_BIN, and the
|
||
// question-preference runtime call still resolves the same way.
|
||
expect(content).toContain('$GSTACK_BIN/gstack-skill-start');
|
||
expect(content).toContain('$GSTACK_BIN/gstack-question-preference');
|
||
// The upgrade-skill doc reference moved into the script's upgrade-flow
|
||
// block, resolved $0-relative ($_ROOT_DIR) — host-neutral by construction,
|
||
// so the Codex render no longer needs its own copy.
|
||
expect(SKILL_START_SCRIPT).toContain('$_ROOT_DIR/gstack-upgrade/SKILL.md');
|
||
expect(SKILL_START_SCRIPT).toContain('_ROOT_DIR=$(dirname "$_BIN")');
|
||
expect(content).not.toContain('~/.codex/skills/gstack/bin/gstack-config get telemetry');
|
||
});
|
||
|
||
// ─── Path rewriting regression tests ─────────────────────────
|
||
|
||
test('sidecar paths resolve through $GSTACK_ROOT (not gstack-review/)', () => {
|
||
// #2518: templates now anchor sidecars at the installed skill root
|
||
// (~/.claude/skills/gstack/review/...), which the codex path rewrite turns
|
||
// into $GSTACK_ROOT/review/... — resolved by the preamble against the
|
||
// repo-local .agents root or the global install. The old repo-relative
|
||
// form (.claude/skills/review/) only resolved inside gstack's own checkout.
|
||
const content = fs.readFileSync(path.join(AGENTS_DIR, 'gstack-review', 'SKILL.md'), 'utf-8');
|
||
expect(content).toContain('$GSTACK_ROOT/review/checklist.md');
|
||
// design-checklist.md is now referenced via Review Army specialist (Claude only, stripped for Codex)
|
||
// Wrong: must NOT reference gstack-review/checklist.md (file doesn't exist there)
|
||
expect(content).not.toContain('.agents/skills/gstack-review/checklist.md');
|
||
});
|
||
|
||
test('sidecar paths in ship skill point to gstack/review/ for pre-landing review', () => {
|
||
const content = fs.readFileSync(path.join(AGENTS_DIR, 'gstack-ship', 'SKILL.md'), 'utf-8');
|
||
// Ship references the review checklist in its pre-landing review step
|
||
if (content.includes('checklist.md')) {
|
||
expect(content).toContain('.agents/skills/gstack/review/');
|
||
expect(content).not.toContain('.agents/skills/gstack-review/checklist');
|
||
}
|
||
});
|
||
|
||
test('greptile-triage sidecar path is correct', () => {
|
||
const content = fs.readFileSync(path.join(AGENTS_DIR, 'gstack-review', 'SKILL.md'), 'utf-8');
|
||
if (content.includes('greptile-triage')) {
|
||
expect(content).toContain('$GSTACK_ROOT/review/greptile-triage.md');
|
||
expect(content).not.toContain('.agents/skills/gstack-review/greptile-triage');
|
||
}
|
||
});
|
||
|
||
test('all four path rewrite rules produce correct output', () => {
|
||
// Test each of the 4 path rewrite rules individually
|
||
const content = fs.readFileSync(path.join(AGENTS_DIR, 'gstack-review', 'SKILL.md'), 'utf-8');
|
||
|
||
// Rule 1: ~/.claude/skills/gstack → $GSTACK_ROOT
|
||
expect(content).not.toContain('~/.claude/skills/gstack');
|
||
expect(content).toContain('$GSTACK_ROOT');
|
||
|
||
// Rule 2: .claude/skills/gstack → .agents/skills/gstack
|
||
expect(content).not.toContain('.claude/skills/gstack');
|
||
|
||
// Rule 3: .claude/skills/review → .agents/skills/gstack/review
|
||
expect(content).not.toContain('.claude/skills/review');
|
||
|
||
// Rule 4: .claude/skills → .agents/skills (catch-all)
|
||
expect(content).not.toContain('.claude/skills');
|
||
});
|
||
|
||
test('path rewrite rules apply to all Codex skills with sidecar references', () => {
|
||
// Verify across ALL generated skills, not just review
|
||
for (const skill of CODEX_SKILLS) {
|
||
const content = fs.readFileSync(path.join(AGENTS_DIR, skill.codexName, 'SKILL.md'), 'utf-8');
|
||
// No skill should reference Claude paths
|
||
expect(content).not.toContain('~/.claude/skills');
|
||
expect(content).not.toContain('.claude/skills');
|
||
if (content.includes('gstack-config') || content.includes('gstack-update-check') || content.includes('gstack-telemetry-log')) {
|
||
expect(content).toContain('$GSTACK_ROOT');
|
||
}
|
||
// If a skill references checklist.md, it must use the correct sidecar path
|
||
if (content.includes('checklist.md') && !content.includes('design-checklist.md')) {
|
||
expect(content).not.toContain('gstack-review/checklist.md');
|
||
}
|
||
}
|
||
});
|
||
|
||
// ─── Claude output regression guard ─────────────────────────
|
||
|
||
test('Claude output uses installed-root review paths (#2518)', () => {
|
||
// Codex changes must NOT affect Claude output; the Claude form is the
|
||
// installed-root anchor, not the old repo-relative path that only
|
||
// resolved inside gstack's own checkout.
|
||
const content = fs.readFileSync(path.join(ROOT, 'review', 'SKILL.md'), 'utf-8');
|
||
expect(content).toContain('~/.claude/skills/gstack/review/checklist.md');
|
||
expect(content).toContain('~/.claude/skills/gstack');
|
||
// Must NOT contain Codex HOST paths. `~/.codex/sessions/` is exempt: the
|
||
// timeout-wrapper guidance documents the Codex CLI's own rollout-log
|
||
// location (a user-facing CLI path, same class as ~/.codex/logs/ in the
|
||
// codex skill), not the gstack Codex host install path.
|
||
// `~/.codex/config.toml` is the same user-facing class: the shared
|
||
// codexPreflight's model_unusable branch (#2477) points at the CLI's own
|
||
// config file, where the rejected `model =` pin lives.
|
||
expect(content).not.toContain('.agents/skills');
|
||
expect(
|
||
content
|
||
.replaceAll('~/.codex/sessions/', '')
|
||
.replaceAll('~/.codex/config.toml', ''),
|
||
).not.toContain('~/.codex/');
|
||
});
|
||
|
||
test('Claude output unchanged: ship skill still uses .claude/skills/ paths', () => {
|
||
const content = readShipUnion();
|
||
expect(content).toContain('~/.claude/skills/gstack');
|
||
expect(content).not.toContain('.agents/skills');
|
||
// ~/.codex/sessions/ is the Codex CLI's rollout-log path (user-facing),
|
||
// documented by the adversarial-pass timeout guidance; ~/.codex/config.toml
|
||
// is the CLI's own config file (model_unusable guidance, #2477) — see the
|
||
// review test above.
|
||
expect(
|
||
content
|
||
.replaceAll('~/.codex/sessions/', '')
|
||
.replaceAll('~/.codex/config.toml', ''),
|
||
).not.toContain('~/.codex/');
|
||
});
|
||
|
||
test('Claude output unchanged: all Claude skills have zero Codex paths', () => {
|
||
for (const skill of CLAUDE_GENERATED_SKILLS) {
|
||
const content = fs.readFileSync(path.join(ROOT, skill.dir, 'SKILL.md'), 'utf-8');
|
||
// pair-agent legitimately documents how Codex agents store credentials.
|
||
// codex + autoplan document the Codex CLI auth file (~/.codex/auth.json)
|
||
// and log path (~/.codex/logs/) — those are user-facing Codex CLI paths,
|
||
// not the gstack Codex host install path. ~/.codex/sessions/ (rollout
|
||
// logs, referenced by the review/ship timeout guidance) and
|
||
// ~/.codex/config.toml (the model_unusable guidance in the shared
|
||
// codexPreflight, #2477) are the same user-facing class, so they are
|
||
// scrubbed before the ban.
|
||
if (skill.dir !== 'pair-agent' && skill.dir !== 'codex' && skill.dir !== 'autoplan') {
|
||
expect(
|
||
content
|
||
.replaceAll('~/.codex/sessions/', '')
|
||
.replaceAll('~/.codex/config.toml', ''),
|
||
).not.toContain('~/.codex/');
|
||
}
|
||
// gstack-upgrade legitimately references .agents/skills for cross-platform detection
|
||
if (skill.dir !== 'gstack-upgrade') {
|
||
expect(content).not.toContain('.agents/skills');
|
||
}
|
||
}
|
||
});
|
||
|
||
// ─── Design outside voices: Codex host guard ─────────────────
|
||
|
||
test('codex host retains design outside voices through Claude Code', () => {
|
||
const codexContent = readExternalSkillUnion(EXTERNAL_OUT, '.agents', 'design-review');
|
||
expect(codexContent).toContain('Design Outside Voices');
|
||
expect(codexContent).toContain('gstack-claude-code');
|
||
});
|
||
|
||
test('codex host does not include Codex design block in ship', () => {
|
||
const codexContent = fs.readFileSync(path.join(AGENTS_DIR, 'gstack-ship', 'SKILL.md'), 'utf-8');
|
||
expect(codexContent).not.toContain('Codex design voice');
|
||
});
|
||
|
||
// ─── Explicit --model override wins over the host default ────
|
||
// Without --model the codex host renders its defaultModel (gpt) — pinned by
|
||
// the golden test. This pins the OTHER direction through the real CLI:
|
||
// `./setup --host codex --model <id>` depends on it. The override renders
|
||
// into its OWN out-dir, so no restore pass is needed — the host-default
|
||
// render (EXTERNAL_OUT) is untouched and asserted directly.
|
||
test('explicit --model overrides the codex host default', () => {
|
||
const overrideOut = fs.mkdtempSync(path.join(os.tmpdir(), 'gstack-model-override-'));
|
||
try {
|
||
const override = Bun.spawnSync(['bun', 'run', 'scripts/gen-skill-docs.ts', '--host', 'codex', '--model', 'claude', '--out-dir', overrideOut], {
|
||
cwd: ROOT,
|
||
stdout: 'pipe',
|
||
stderr: 'pipe',
|
||
timeout: 120_000,
|
||
});
|
||
expect(override.exitCode).toBe(0);
|
||
const content = fs.readFileSync(path.join(overrideOut, '.agents', 'skills', 'gstack-ship', 'SKILL.md'), 'utf-8');
|
||
expect(content).toContain('Model-Specific Behavioral Patch (claude)');
|
||
// The overlay now travels as --model into gstack-skill-start, which
|
||
// echoes MODEL_OVERLAY at runtime.
|
||
expect(content).toContain('--model "claude"');
|
||
} finally {
|
||
fs.rmSync(overrideOut, { recursive: true, force: true });
|
||
}
|
||
// Host-default direction: the untouched EXTERNAL_OUT render carries gpt.
|
||
const hostDefault = fs.readFileSync(path.join(AGENTS_DIR, 'gstack-ship', 'SKILL.md'), 'utf-8');
|
||
expect(hostDefault).toContain('Model-Specific Behavioral Patch (gpt)');
|
||
expect(hostDefault).toContain('--model "gpt"');
|
||
});
|
||
});
|
||
|
||
// ─── Factory generation tests ────────────────────────────────
|
||
|
||
describe('Factory generation (--host factory)', () => {
|
||
// .factory/ is gitignored — read the module-level out-dir render
|
||
// (--host all covers factory) instead of regenerating in place.
|
||
const FACTORY_DIR = path.join(EXTERNAL_OUT, '.factory', 'skills');
|
||
|
||
// Fresh out-dir → the vendored-dev-mode symlink loop can never occur, so
|
||
// every template renders (see the Codex discovery note above).
|
||
const FACTORY_SKILLS = (() => {
|
||
const skills: Array<{ dir: string; factoryName: string }> = [];
|
||
if (fs.existsSync(path.join(ROOT, 'SKILL.md.tmpl'))) {
|
||
skills.push({ dir: '.', factoryName: 'gstack' });
|
||
}
|
||
for (const entry of fs.readdirSync(ROOT, { withFileTypes: true })) {
|
||
if (!entry.isDirectory() || entry.name.startsWith('.') || entry.name === 'node_modules') continue;
|
||
if (entry.name === 'codex') continue;
|
||
if (!fs.existsSync(path.join(ROOT, entry.name, 'SKILL.md.tmpl'))) continue;
|
||
const factoryName = entry.name.startsWith('gstack-') ? entry.name : `gstack-${entry.name}`;
|
||
skills.push({ dir: entry.name, factoryName });
|
||
}
|
||
return skills;
|
||
})();
|
||
|
||
test('--host factory generates correct output paths', () => {
|
||
for (const skill of FACTORY_SKILLS) {
|
||
const skillMd = path.join(FACTORY_DIR, skill.factoryName, 'SKILL.md');
|
||
expect(fs.existsSync(skillMd)).toBe(true);
|
||
}
|
||
});
|
||
|
||
test('Factory frontmatter has name + description + user-invocable', () => {
|
||
for (const skill of FACTORY_SKILLS) {
|
||
const content = fs.readFileSync(path.join(FACTORY_DIR, skill.factoryName, 'SKILL.md'), 'utf-8');
|
||
const fmEnd = content.indexOf('\n---', 4);
|
||
const frontmatter = content.slice(4, fmEnd);
|
||
expect(frontmatter).toContain('name:');
|
||
expect(frontmatter).toContain('description:');
|
||
expect(frontmatter).toContain('user-invocable: true');
|
||
expect(frontmatter).not.toContain('allowed-tools:');
|
||
expect(frontmatter).not.toContain('preamble-tier:');
|
||
expect(frontmatter).not.toContain('sensitive:');
|
||
}
|
||
});
|
||
|
||
test('sensitive skills have disable-model-invocation', () => {
|
||
const SENSITIVE = ['gstack-ship', 'gstack-land-and-deploy', 'gstack-guard', 'gstack-careful', 'gstack-freeze', 'gstack-unfreeze'];
|
||
for (const name of SENSITIVE) {
|
||
const content = fs.readFileSync(path.join(FACTORY_DIR, name, 'SKILL.md'), 'utf-8');
|
||
const fmEnd = content.indexOf('\n---', 4);
|
||
const frontmatter = content.slice(4, fmEnd);
|
||
expect(frontmatter).toContain('disable-model-invocation: true');
|
||
}
|
||
});
|
||
|
||
test('non-sensitive skills lack disable-model-invocation', () => {
|
||
const NON_SENSITIVE = ['gstack-qa', 'gstack-review', 'gstack-investigate', 'gstack-browse'];
|
||
for (const name of NON_SENSITIVE) {
|
||
const content = fs.readFileSync(path.join(FACTORY_DIR, name, 'SKILL.md'), 'utf-8');
|
||
const fmEnd = content.indexOf('\n---', 4);
|
||
const frontmatter = content.slice(4, fmEnd);
|
||
expect(frontmatter).not.toContain('disable-model-invocation');
|
||
}
|
||
});
|
||
|
||
test('no .claude/skills/ in Factory output', () => {
|
||
for (const skill of FACTORY_SKILLS) {
|
||
const content = fs.readFileSync(path.join(FACTORY_DIR, skill.factoryName, 'SKILL.md'), 'utf-8');
|
||
expect(content).not.toContain('.claude/skills');
|
||
}
|
||
});
|
||
|
||
test('no ~/.claude/skills/ paths in Factory output', () => {
|
||
for (const skill of FACTORY_SKILLS) {
|
||
const content = fs.readFileSync(path.join(FACTORY_DIR, skill.factoryName, 'SKILL.md'), 'utf-8');
|
||
// ~/.claude/skills should be rewritten, but ~/.claude/plans is legitimate
|
||
// (plan directory lookup) and ~/.claude/ in codex prompts is intentional
|
||
expect(content).not.toContain('~/.claude/skills');
|
||
}
|
||
});
|
||
|
||
test('Factory installs both outside-review wrapper skills', () => {
|
||
for (const skill of ['gstack-codex', 'gstack-claude-code']) {
|
||
expect(fs.existsSync(path.join(FACTORY_DIR, skill, 'SKILL.md'))).toBe(true);
|
||
}
|
||
});
|
||
|
||
test('Factory keeps Codex integration blocks', () => {
|
||
// Factory users CAN use Codex second opinions (codex exec is a standalone binary)
|
||
const shipContent = fs.readFileSync(path.join(FACTORY_DIR, 'gstack-ship', 'SKILL.md'), 'utf-8');
|
||
expect(shipContent).toContain('codex');
|
||
});
|
||
|
||
test('no agents/openai.yaml in Factory output', () => {
|
||
for (const skill of FACTORY_SKILLS) {
|
||
const yamlPath = path.join(FACTORY_DIR, skill.factoryName, 'agents', 'openai.yaml');
|
||
expect(fs.existsSync(yamlPath)).toBe(false);
|
||
}
|
||
});
|
||
|
||
test('--host droid alias works', () => {
|
||
const factoryResult = Bun.spawnSync(['bun', 'run', 'scripts/gen-skill-docs.ts', '--host', 'factory', '--dry-run', '--out-dir', EXTERNAL_OUT], {
|
||
cwd: ROOT, stdout: 'pipe', stderr: 'pipe', timeout: 120_000,
|
||
});
|
||
const droidResult = Bun.spawnSync(['bun', 'run', 'scripts/gen-skill-docs.ts', '--host', 'droid', '--dry-run', '--out-dir', EXTERNAL_OUT], {
|
||
cwd: ROOT, stdout: 'pipe', stderr: 'pipe', timeout: 120_000,
|
||
});
|
||
expect(factoryResult.exitCode).toBe(0);
|
||
expect(droidResult.exitCode).toBe(0);
|
||
expect(factoryResult.stdout.toString()).toBe(droidResult.stdout.toString());
|
||
});
|
||
|
||
test('--host factory --dry-run freshness', () => {
|
||
const result = Bun.spawnSync(['bun', 'run', 'scripts/gen-skill-docs.ts', '--host', 'factory', '--dry-run', '--out-dir', EXTERNAL_OUT], {
|
||
cwd: ROOT, stdout: 'pipe', stderr: 'pipe', timeout: 120_000,
|
||
});
|
||
expect(result.exitCode).toBe(0);
|
||
const output = result.stdout.toString();
|
||
for (const skill of FACTORY_SKILLS) {
|
||
expect(output).toContain(`FRESH: .factory/skills/${skill.factoryName}/SKILL.md`);
|
||
}
|
||
expect(output).not.toContain('STALE');
|
||
});
|
||
|
||
test('Factory preamble uses .factory paths', () => {
|
||
const content = fs.readFileSync(path.join(FACTORY_DIR, 'gstack-review', 'SKILL.md'), 'utf-8');
|
||
expect(content).toContain('GSTACK_ROOT');
|
||
expect(content).toContain('$_ROOT/.factory/skills/gstack');
|
||
expect(content).toContain('$GSTACK_BIN/gstack-config');
|
||
});
|
||
});
|
||
|
||
// ─── Parameterized host smoke tests (config-driven) ─────────
|
||
|
||
import { ALL_HOST_CONFIGS, getExternalHosts } from '../hosts/index';
|
||
import { sliceBetween } from './helpers/skill-fixture';
|
||
|
||
describe('Parameterized host smoke tests', () => {
|
||
// Every external host was rendered up front by the module-level
|
||
// `--host all --out-dir EXTERNAL_OUT` render, so the per-host `--dry-run`
|
||
// freshness checks are deterministic: they compare a regeneration against
|
||
// that render — an idempotency/determinism check that catches
|
||
// non-deterministic gen without ever writing (or depending on) the live
|
||
// gitignored host dirs. The tracked-claude freshness test
|
||
// (`generated files are fresh`) runs earlier and is unaffected.
|
||
for (const hostConfig of getExternalHosts()) {
|
||
describe(`${hostConfig.displayName} (--host ${hostConfig.name})`, () => {
|
||
const hostDir = path.join(EXTERNAL_OUT, hostConfig.hostSubdir, 'skills');
|
||
|
||
test('generates output that exists on disk', () => {
|
||
// The module-level --host all render must have produced this host's tree.
|
||
expect(fs.existsSync(hostDir)).toBe(true);
|
||
const skills = fs.readdirSync(hostDir).filter(d =>
|
||
fs.existsSync(path.join(hostDir, d, 'SKILL.md'))
|
||
);
|
||
expect(skills.length).toBeGreaterThan(0);
|
||
});
|
||
|
||
test('no .claude/skills path leakage outside repo-root sidecar symlinks', () => {
|
||
if (!fs.existsSync(hostDir)) return; // skip if not generated
|
||
const skills = fs.readdirSync(hostDir);
|
||
for (const skill of skills) {
|
||
// Dev installs may mount the repo root at host/skills/gstack as a runtime
|
||
// sidecar. The generator skips that symlink loop, so leakage checks should too.
|
||
if (isRepoRootSymlink(path.join(hostDir, skill))) continue;
|
||
const skillMd = path.join(hostDir, skill, 'SKILL.md');
|
||
if (!fs.existsSync(skillMd)) continue;
|
||
const content = fs.readFileSync(skillMd, 'utf-8');
|
||
const leaks = externalHostPathLeaks(content);
|
||
if (leaks.length > 0) {
|
||
throw new Error(`${skill}: .claude/skills leakage:\n${leaks.slice(0, 3).join('\n')}`);
|
||
}
|
||
}
|
||
});
|
||
|
||
test('frontmatter has name and description', () => {
|
||
if (!fs.existsSync(hostDir)) return;
|
||
const skills = fs.readdirSync(hostDir);
|
||
for (const skill of skills) {
|
||
const skillMd = path.join(hostDir, skill, 'SKILL.md');
|
||
if (!fs.existsSync(skillMd)) continue;
|
||
const content = fs.readFileSync(skillMd, 'utf-8');
|
||
expect(content).toMatch(/^---\n/);
|
||
expect(content).toMatch(/^name:\s/m);
|
||
expect(content).toMatch(/^description:\s/m);
|
||
}
|
||
});
|
||
|
||
test('--dry-run freshness check passes', () => {
|
||
const result = Bun.spawnSync(
|
||
['bun', 'run', 'scripts/gen-skill-docs.ts', '--host', hostConfig.name, '--dry-run', '--out-dir', EXTERNAL_OUT],
|
||
{ cwd: ROOT, stdout: 'pipe', stderr: 'pipe', timeout: 120_000 }
|
||
);
|
||
expect(result.exitCode).toBe(0);
|
||
const output = result.stdout.toString();
|
||
expect(output).not.toContain('STALE');
|
||
});
|
||
|
||
if (hostConfig.generation.skipSkills?.includes('codex')) {
|
||
test('/codex skill excluded', () => {
|
||
expect(fs.existsSync(path.join(hostDir, 'gstack-codex', 'SKILL.md'))).toBe(false);
|
||
});
|
||
}
|
||
});
|
||
}
|
||
});
|
||
|
||
// ─── --host all tests ────────────────────────────────────────
|
||
|
||
describe('--host all', () => {
|
||
// Same determinism guard as the parameterized block: the module-level
|
||
// `--host all --out-dir EXTERNAL_OUT` render is the comparison baseline, so
|
||
// this dry-run reports FRESH regardless of live-tree state — and proves the
|
||
// claude host plus every external host regenerate deterministically.
|
||
test('--host all generates for all registered hosts', () => {
|
||
const result = Bun.spawnSync(['bun', 'run', 'scripts/gen-skill-docs.ts', '--host', 'all', '--dry-run', '--out-dir', EXTERNAL_OUT], {
|
||
cwd: ROOT, stdout: 'pipe', stderr: 'pipe', timeout: 120_000,
|
||
});
|
||
expect(result.exitCode).toBe(0);
|
||
const output = result.stdout.toString();
|
||
// All hosts should appear in output
|
||
expect(output).toContain('FRESH: SKILL.md'); // claude
|
||
for (const hostConfig of getExternalHosts()) {
|
||
expect(output).toContain(`FRESH: ${hostConfig.hostSubdir}/skills/`);
|
||
}
|
||
});
|
||
});
|
||
|
||
// ─── Setup script validation ─────────────────────────────────
|
||
// These tests verify the setup script's install layout matches
|
||
// what the generator produces — catching the bug where setup
|
||
// installed Claude-format source dirs for Codex users.
|
||
|
||
describe('setup script validation', () => {
|
||
const setupContent = fs.readFileSync(path.join(ROOT, 'setup'), 'utf-8');
|
||
|
||
test('setup has separate link functions for Claude and Codex', () => {
|
||
expect(setupContent).toContain('link_claude_skill_dirs');
|
||
expect(setupContent).toContain('link_codex_skill_dirs');
|
||
// Old unified function must not exist
|
||
expect(setupContent).not.toMatch(/^link_skill_dirs\(\)/m);
|
||
});
|
||
|
||
test('Claude install uses link_claude_skill_dirs', () => {
|
||
// The Claude install section (section 4) should use the Claude function
|
||
const claudeSection = setupContent.slice(
|
||
setupContent.indexOf('# 4. Install for Claude'),
|
||
setupContent.indexOf('# 5. Install for Codex')
|
||
);
|
||
expect(claudeSection).toContain('link_claude_skill_dirs');
|
||
expect(claudeSection).not.toContain('link_codex_skill_dirs');
|
||
});
|
||
|
||
test('Codex install uses link_codex_skill_dirs', () => {
|
||
// The Codex install section (section 5) should use the Codex function
|
||
// End marker: the next numbered section header (a marker that doesn't
|
||
// exist slices to EOF and the assertion reads unrelated sections).
|
||
const codexSection = setupContent.slice(
|
||
setupContent.indexOf('# 5. Install for Codex'),
|
||
setupContent.indexOf('# 6. Install for Kiro')
|
||
);
|
||
expect(setupContent.indexOf('# 6. Install for Kiro')).toBeGreaterThan(-1);
|
||
expect(codexSection).toContain('create_codex_runtime_root');
|
||
expect(codexSection).toContain('link_codex_skill_dirs');
|
||
expect(codexSection).not.toContain('link_claude_skill_dirs');
|
||
expect(codexSection).not.toContain('_link_or_copy "$GSTACK_DIR" "$CODEX_GSTACK"');
|
||
});
|
||
|
||
test('Codex install prefers repo-local .agents/skills when setup runs from there', () => {
|
||
expect(setupContent).toContain('SKILLS_PARENT_BASENAME');
|
||
expect(setupContent).toContain('CODEX_REPO_LOCAL=0');
|
||
expect(setupContent).toContain('[ "$SKILLS_PARENT_BASENAME" = ".agents" ]');
|
||
expect(setupContent).toContain('CODEX_REPO_LOCAL=1');
|
||
expect(setupContent).toContain('CODEX_SKILLS="$INSTALL_SKILLS_DIR"');
|
||
});
|
||
|
||
test('setup separates install path from source path for symlinked repo-local installs', () => {
|
||
expect(setupContent).toContain('INSTALL_GSTACK_DIR=');
|
||
expect(setupContent).toContain('SOURCE_GSTACK_DIR=');
|
||
expect(setupContent).toContain('INSTALL_SKILLS_DIR=');
|
||
expect(setupContent).toContain('CODEX_GSTACK="$INSTALL_GSTACK_DIR"');
|
||
expect(setupContent).toContain('link_codex_skill_dirs "$SOURCE_GSTACK_DIR" "$CODEX_SKILLS"');
|
||
});
|
||
|
||
test('Codex installs always create sidecar runtime assets for the real skill target', () => {
|
||
expect(setupContent).toContain('if [ "$INSTALL_CODEX" -eq 1 ]; then');
|
||
expect(setupContent).toContain('create_agents_sidecar "$SOURCE_GSTACK_DIR"');
|
||
});
|
||
|
||
test('link_codex_skill_dirs reads from .agents/skills/', () => {
|
||
// The Codex link function must reference .agents/skills for generated Codex skills
|
||
const fnStart = setupContent.indexOf('link_codex_skill_dirs()');
|
||
const fnEnd = setupContent.indexOf('}', setupContent.indexOf('linked[@]}', fnStart));
|
||
const fnBody = setupContent.slice(fnStart, fnEnd);
|
||
expect(fnBody).toContain('.agents/skills');
|
||
expect(fnBody).toContain('gstack*');
|
||
});
|
||
|
||
test('link_claude_skill_dirs creates real directories with absolute SKILL.md symlinks', () => {
|
||
// Claude links should be real directories with absolute SKILL.md symlinks
|
||
// to ensure Claude Code discovers them as top-level skills (not nested under gstack/)
|
||
const fnStart = setupContent.indexOf('link_claude_skill_dirs()');
|
||
const fnEnd = setupContent.indexOf('}', setupContent.indexOf('linked[@]}', fnStart));
|
||
const fnBody = setupContent.slice(fnStart, fnEnd);
|
||
expect(fnBody).toContain('mkdir -p "$target"');
|
||
// v1.36.0.0: routes through _link_or_copy helper for Windows fallback (cp on MSYS2/Git Bash).
|
||
// v1.67 (#2569): the source is render-aware — canonical SKILL.md, or the
|
||
// rendered :user variant from ${GSTACK_HOME}/render/claude when present.
|
||
expect(fnBody).toContain('_skill_md_src="$gstack_dir/$dir_name/SKILL.md"');
|
||
expect(fnBody).toContain('_link_or_copy "$_skill_md_src" "$target/SKILL.md"');
|
||
});
|
||
|
||
// REGRESSION: cleanup functions must handle both old symlinks AND new real-directory pattern
|
||
test('cleanup functions handle real directories with symlinked SKILL.md', () => {
|
||
// cleanup_old_claude_symlinks must detect and remove real dirs with SKILL.md symlinks
|
||
const cleanupOldStart = setupContent.indexOf('cleanup_old_claude_symlinks()');
|
||
const cleanupOldEnd = setupContent.indexOf('}', setupContent.indexOf('cleaned up old', cleanupOldStart));
|
||
const cleanupOldBody = setupContent.slice(cleanupOldStart, cleanupOldEnd);
|
||
expect(cleanupOldBody).toContain('-d "$old_target"');
|
||
expect(cleanupOldBody).toContain('-L "$old_target/SKILL.md"');
|
||
expect(cleanupOldBody).toContain('rm -rf "$old_target"');
|
||
|
||
// cleanup_prefixed_claude_symlinks must also handle the new pattern
|
||
const cleanupPrefixedStart = setupContent.indexOf('cleanup_prefixed_claude_symlinks()');
|
||
const cleanupPrefixedEnd = setupContent.indexOf('}', setupContent.indexOf('cleaned up prefixed', cleanupPrefixedStart));
|
||
const cleanupPrefixedBody = setupContent.slice(cleanupPrefixedStart, cleanupPrefixedEnd);
|
||
expect(cleanupPrefixedBody).toContain('-d "$prefixed_target"');
|
||
expect(cleanupPrefixedBody).toContain('-L "$prefixed_target/SKILL.md"');
|
||
expect(cleanupPrefixedBody).toContain('rm -rf "$prefixed_target"');
|
||
});
|
||
|
||
// REGRESSION: link function must upgrade old directory symlinks
|
||
test('link_claude_skill_dirs removes old directory symlinks before creating real dirs', () => {
|
||
const fnStart = setupContent.indexOf('link_claude_skill_dirs()');
|
||
const fnEnd = setupContent.indexOf('}', setupContent.indexOf('linked[@]}', fnStart));
|
||
const fnBody = setupContent.slice(fnStart, fnEnd);
|
||
// Must check for and remove old symlinks before mkdir
|
||
expect(fnBody).toContain('if [ -L "$target" ]');
|
||
expect(fnBody).toContain('rm -f "$target"');
|
||
});
|
||
|
||
test('setup links root gstack skill through a thin Claude wrapper alias', () => {
|
||
const fnStart = setupContent.indexOf('link_claude_root_skill_alias()');
|
||
const fnEnd = setupContent.indexOf('# ─── Helper: remove old unprefixed Claude skill entries', fnStart);
|
||
const fnBody = setupContent.slice(fnStart, fnEnd);
|
||
expect(fnBody).toContain('_gstack-command');
|
||
// #2511: the alias must be a rewritten COPY (unique frontmatter name),
|
||
// never a verbatim symlink of the canonical SKILL.md.
|
||
expect(fnBody).toContain('_install_alias_skill_md "$gstack_dir/SKILL.md" "$target" "_gstack-command"');
|
||
expect(fnBody).not.toContain('_link_or_copy "$gstack_dir/SKILL.md"');
|
||
|
||
const claudeSection = setupContent.slice(
|
||
setupContent.indexOf('# 4. Install for Claude'),
|
||
setupContent.indexOf('# 5. Install for Codex')
|
||
);
|
||
expect(claudeSection).toContain('link_claude_root_skill_alias "$SOURCE_GSTACK_DIR" "$INSTALL_SKILLS_DIR"');
|
||
});
|
||
|
||
test('setup supports --host auto|claude|codex|kiro|opencode|cursor; slate is informational', () => {
|
||
expect(setupContent).toContain('--host');
|
||
// #2361: slate moved OUT of the install accept-list (it was accepted but
|
||
// never dispatched — a silent exit-0 no-op) into an informational arm.
|
||
expect(setupContent).toContain('claude|codex|kiro|factory|opencode|cursor|auto');
|
||
expect(setupContent).toMatch(/^ {2}slate\)/m);
|
||
});
|
||
|
||
test('auto mode detects claude, codex, kiro, and opencode binaries', () => {
|
||
expect(setupContent).toContain('command -v claude');
|
||
expect(setupContent).toContain('command -v codex');
|
||
expect(setupContent).toContain('command -v kiro-cli');
|
||
expect(setupContent).toContain('command -v opencode');
|
||
});
|
||
|
||
// T1: Sidecar skip guard — prevents .agents/skills/gstack from being linked as a skill
|
||
test('link_codex_skill_dirs skips the gstack sidecar directory', () => {
|
||
const fnStart = setupContent.indexOf('link_codex_skill_dirs()');
|
||
const fnEnd = setupContent.indexOf('}', setupContent.indexOf('done', fnStart));
|
||
const fnBody = setupContent.slice(fnStart, fnEnd);
|
||
expect(fnBody).toContain('[ "$skill_name" = "gstack" ] && continue');
|
||
});
|
||
|
||
// T2: Dynamic $GSTACK_ROOT paths in generated Codex preambles
|
||
test('generated Codex preambles use dynamic GSTACK_ROOT paths', () => {
|
||
// Read the module-level out-dir render (always present).
|
||
const codexSkillDir = path.join(EXTERNAL_OUT, '.agents', 'skills', 'gstack-ship');
|
||
const content = fs.readFileSync(path.join(codexSkillDir, 'SKILL.md'), 'utf-8');
|
||
expect(content).toContain('GSTACK_ROOT=');
|
||
expect(content).toContain('$GSTACK_BIN/');
|
||
});
|
||
|
||
test('setup installs Kiro-native output and runtime assets', () => {
|
||
expect(setupContent).toContain('INSTALL_KIRO=');
|
||
expect(setupContent).toContain('kiro-cli');
|
||
expect(setupContent).toContain('KIRO_SKILLS=');
|
||
expect(setupContent).toContain('KIRO_SKILLS="$HOME/.kiro/skills"');
|
||
expect(setupContent).toContain('KIRO_DIR="$SOURCE_GSTACK_DIR/.kiro/skills"');
|
||
expect(setupContent).toContain('gen:skill-docs --host kiro');
|
||
expect(setupContent).toContain('$KIRO_GSTACK/lib');
|
||
});
|
||
|
||
test('setup supports --host opencode with install section and OpenCode skill path vars', () => {
|
||
expect(setupContent).toContain('INSTALL_OPENCODE=');
|
||
expect(setupContent).toContain('OPENCODE_SKILLS="$HOME/.config/opencode/skills"');
|
||
expect(setupContent).toContain('OPENCODE_GSTACK="$OPENCODE_SKILLS/gstack"');
|
||
});
|
||
|
||
// --host cursor full install slice (#1358, PR #2547 by @szsunyuan re-derived)
|
||
test('auto mode detects Cursor via binary or ~/.cursor directory', () => {
|
||
expect(setupContent).toContain('command -v cursor');
|
||
expect(setupContent).toContain('[ -d "$HOME/.cursor" ] && INSTALL_CURSOR=1');
|
||
});
|
||
|
||
test('setup supports --host cursor with install section and Cursor skill path vars', () => {
|
||
expect(setupContent).toContain('INSTALL_CURSOR=');
|
||
expect(setupContent).toContain('CURSOR_SKILLS="$HOME/.cursor/skills"');
|
||
expect(setupContent).toContain('CURSOR_GSTACK="$CURSOR_SKILLS/gstack"');
|
||
expect(setupContent).toContain('create_cursor_runtime_root');
|
||
expect(setupContent).toContain('create_cursor_sidecar');
|
||
expect(setupContent).toContain('link_cursor_skill_dirs');
|
||
expect(setupContent).toContain('gstack ready (cursor).');
|
||
});
|
||
|
||
test('create_cursor_runtime_root exposes only Cursor runtime assets', () => {
|
||
const fnStart = setupContent.indexOf('create_cursor_runtime_root()');
|
||
const fnEnd = setupContent.indexOf('create_cursor_sidecar()', fnStart);
|
||
const fnBody = setupContent.slice(fnStart, fnEnd);
|
||
expect(fnBody).toContain('gstack/SKILL.md');
|
||
expect(fnBody).toContain('browse/dist');
|
||
expect(fnBody).toContain('browse/bin');
|
||
expect(fnBody).toContain('gstack-upgrade/SKILL.md');
|
||
expect(fnBody).toContain('checklist.md');
|
||
expect(fnBody).toContain('TODOS-format.md');
|
||
// bin scripts import ../lib — the two must travel together.
|
||
expect(fnBody).toContain('$cursor_gstack/lib');
|
||
expect(fnBody).not.toContain('design-checklist.md');
|
||
expect(fnBody).not.toContain('greptile-triage.md');
|
||
expect(fnBody).not.toContain('review/specialists');
|
||
expect(fnBody).not.toContain('qa/templates');
|
||
expect(fnBody).not.toContain('_link_or_copy "$gstack_dir" "$cursor_gstack"');
|
||
});
|
||
|
||
test('create_cursor_sidecar plants runtime assets without wiping generated SKILL.md', () => {
|
||
const fnStart = setupContent.indexOf('create_cursor_sidecar()');
|
||
const fnEnd = setupContent.indexOf('link_cursor_skill_dirs()', fnStart);
|
||
const fnBody = setupContent.slice(fnStart, fnEnd);
|
||
expect(fnBody).toContain('.cursor/skills/gstack');
|
||
expect(fnBody).toContain('bin');
|
||
expect(fnBody).toContain('browse/dist');
|
||
expect(fnBody).toContain('browse/bin');
|
||
expect(fnBody).toContain('ETHOS.md');
|
||
expect(fnBody).not.toContain('rm -rf');
|
||
});
|
||
|
||
test('link_cursor_skill_dirs skips the gstack runtime root directory', () => {
|
||
const fnStart = setupContent.indexOf('link_cursor_skill_dirs()');
|
||
const fnEnd = setupContent.indexOf('}', setupContent.indexOf('linked[@]', fnStart));
|
||
const fnBody = setupContent.slice(fnStart, fnEnd);
|
||
expect(fnBody).toContain('[ "$skill_name" = "gstack" ] && continue');
|
||
// #2444-aware guard: Windows bypass, else only replace symlink-or-missing.
|
||
expect(fnBody).toContain('[ "$IS_WINDOWS" -eq 1 ] || [ -L "$target" ] || [ ! -e "$target" ]');
|
||
});
|
||
|
||
// #2142 deleted existing ~/.cursor/skills/<name> dirs with `rm -rf "$target"`
|
||
// before relinking. That can wipe unowned Cursor skills. Only replace a
|
||
// symlink or a missing path; never the whole skills directory.
|
||
test('link_cursor_skill_dirs does not delete unowned Cursor skill directories', () => {
|
||
const fnStart = setupContent.indexOf('link_cursor_skill_dirs()');
|
||
const fnEnd = setupContent.indexOf('}', setupContent.indexOf('linked[@]', fnStart));
|
||
const fnBody = setupContent.slice(fnStart, fnEnd);
|
||
expect(fnBody).not.toContain('rm -rf "$target"');
|
||
expect(fnBody).not.toContain('rm -rf "$skills_dir"');
|
||
expect(setupContent).not.toContain('rm -rf "$CURSOR_SKILLS"');
|
||
});
|
||
|
||
test('Cursor install links generated skills before planting the sidecar', () => {
|
||
const cursorInstall = setupContent.slice(
|
||
setupContent.indexOf('# 6d. Install for Cursor'),
|
||
setupContent.indexOf('# 7. Create .agents/ sidecar'),
|
||
);
|
||
const linkCall = cursorInstall.indexOf('link_cursor_skill_dirs "$SOURCE_GSTACK_DIR"');
|
||
const sidecarCall = cursorInstall.indexOf('create_cursor_sidecar "$SOURCE_GSTACK_DIR"');
|
||
expect(linkCall).toBeGreaterThan(-1);
|
||
expect(sidecarCall).toBeGreaterThan(-1);
|
||
expect(linkCall).toBeLessThan(sidecarCall);
|
||
});
|
||
|
||
test('setup installs OpenCode skills into a nested gstack runtime root', () => {
|
||
expect(setupContent).toContain('create_opencode_runtime_root');
|
||
expect(setupContent).toContain('.opencode/skills');
|
||
expect(setupContent).toContain('review/specialists');
|
||
expect(setupContent).toContain('qa/templates');
|
||
expect(setupContent).toContain('qa/references');
|
||
expect(setupContent).toContain('dx-hall-of-fame.md');
|
||
expect(setupContent).toContain('$opencode_gstack/lib');
|
||
});
|
||
|
||
test('create_agents_sidecar links runtime assets', () => {
|
||
// Sidecar must link bin with its shared lib modules, plus browse, review, qa
|
||
const fnStart = setupContent.indexOf('create_agents_sidecar()');
|
||
const fnEnd = setupContent.indexOf('}', setupContent.indexOf('done', fnStart));
|
||
const fnBody = setupContent.slice(fnStart, fnEnd);
|
||
expect(fnBody).toContain('bin');
|
||
expect(fnBody).toContain('lib');
|
||
expect(fnBody).toContain('browse');
|
||
expect(fnBody).toContain('review');
|
||
expect(fnBody).toContain('qa');
|
||
});
|
||
|
||
test('create_codex_runtime_root exposes only runtime assets', () => {
|
||
const fnStart = setupContent.indexOf('create_codex_runtime_root()');
|
||
const fnEnd = setupContent.indexOf('}', setupContent.indexOf('done', setupContent.indexOf('review/', fnStart)));
|
||
const fnBody = setupContent.slice(fnStart, fnEnd);
|
||
expect(fnBody).toContain('gstack/SKILL.md');
|
||
expect(fnBody).toContain('$codex_gstack/lib');
|
||
expect(fnBody).toContain('browse/dist');
|
||
expect(fnBody).toContain('browse/bin');
|
||
expect(fnBody).toContain('gstack-upgrade/SKILL.md');
|
||
// Review runtime assets (individual files, not the whole dir)
|
||
expect(fnBody).toContain('checklist.md');
|
||
expect(fnBody).toContain('design-checklist.md');
|
||
expect(fnBody).toContain('greptile-triage.md');
|
||
expect(fnBody).toContain('TODOS-format.md');
|
||
expect(fnBody).not.toContain('_link_or_copy "$gstack_dir" "$codex_gstack"');
|
||
});
|
||
|
||
test('create_factory_runtime_root links shared lib modules beside bin', () => {
|
||
const fnStart = setupContent.indexOf('create_factory_runtime_root()');
|
||
const fnEnd = setupContent.indexOf('create_opencode_runtime_root()', fnStart);
|
||
const fnBody = setupContent.slice(fnStart, fnEnd);
|
||
expect(fnBody).toContain('$factory_gstack/bin');
|
||
expect(fnBody).toContain('$factory_gstack/lib');
|
||
});
|
||
|
||
test('direct Codex installs are migrated out of ~/.codex/skills/gstack', () => {
|
||
expect(setupContent).toContain('migrate_direct_codex_install');
|
||
expect(setupContent).toContain('$HOME/.gstack/repos/gstack');
|
||
expect(setupContent).toContain('avoid duplicate skill discovery');
|
||
});
|
||
|
||
// --- Symlink prefix tests (PR #503) ---
|
||
|
||
test('link_claude_skill_dirs applies gstack- prefix by default', () => {
|
||
const fnStart = setupContent.indexOf('link_claude_skill_dirs()');
|
||
const fnEnd = setupContent.indexOf('}', setupContent.indexOf('linked[@]}', fnStart));
|
||
const fnBody = setupContent.slice(fnStart, fnEnd);
|
||
expect(fnBody).toContain('SKILL_PREFIX');
|
||
expect(fnBody).toContain('link_name="gstack-$skill_name"');
|
||
});
|
||
|
||
test('link_claude_skill_dirs preserves already-prefixed dirs', () => {
|
||
const fnStart = setupContent.indexOf('link_claude_skill_dirs()');
|
||
const fnEnd = setupContent.indexOf('}', setupContent.indexOf('linked[@]}', fnStart));
|
||
const fnBody = setupContent.slice(fnStart, fnEnd);
|
||
// gstack-* dirs should keep their name (e.g., gstack-upgrade stays gstack-upgrade)
|
||
expect(fnBody).toContain('gstack-*) link_name="$skill_name"');
|
||
});
|
||
|
||
test('setup supports --no-prefix flag', () => {
|
||
expect(setupContent).toContain('--no-prefix');
|
||
expect(setupContent).toContain('SKILL_PREFIX=0');
|
||
});
|
||
|
||
test('cleanup_old_claude_symlinks removes only gstack-pointing symlinks', () => {
|
||
expect(setupContent).toContain('cleanup_old_claude_symlinks');
|
||
const fnStart = setupContent.indexOf('cleanup_old_claude_symlinks()');
|
||
const fnEnd = setupContent.indexOf('}', setupContent.indexOf('removed[@]}', fnStart));
|
||
const fnBody = setupContent.slice(fnStart, fnEnd);
|
||
// Should check readlink before removing
|
||
expect(fnBody).toContain('readlink');
|
||
expect(fnBody).toContain('gstack/*');
|
||
// Should skip already-prefixed dirs
|
||
expect(fnBody).toContain('gstack-*) continue');
|
||
});
|
||
|
||
test('cleanup runs before link when prefix is enabled', () => {
|
||
// In the Claude install section, cleanup should happen before linking
|
||
const claudeInstallSection = setupContent.slice(
|
||
setupContent.indexOf('INSTALL_CLAUDE'),
|
||
setupContent.lastIndexOf('link_claude_skill_dirs')
|
||
);
|
||
expect(claudeInstallSection).toContain('cleanup_old_claude_symlinks');
|
||
});
|
||
|
||
// --- Persistent config + interactive prompt tests ---
|
||
|
||
test('setup reads skill_prefix from config', () => {
|
||
expect(setupContent).toContain('get skill_prefix');
|
||
expect(setupContent).toContain('GSTACK_CONFIG');
|
||
});
|
||
|
||
test('setup supports --prefix flag', () => {
|
||
expect(setupContent).toContain('--prefix)');
|
||
expect(setupContent).toContain('SKILL_PREFIX=1; SKILL_PREFIX_FLAG=1');
|
||
});
|
||
|
||
test('--prefix and --no-prefix persist to config', () => {
|
||
expect(setupContent).toContain('set skill_prefix');
|
||
});
|
||
|
||
test('interactive prompt shows when no config', () => {
|
||
expect(setupContent).toContain('Short names');
|
||
expect(setupContent).toContain('Namespaced');
|
||
expect(setupContent).toContain('Choice [1/2]');
|
||
});
|
||
|
||
test('non-TTY defaults to flat names', () => {
|
||
// Should check if stdin is a TTY before prompting
|
||
expect(setupContent).toContain('-t 0');
|
||
});
|
||
|
||
test('cleanup_prefixed_claude_symlinks exists and uses readlink', () => {
|
||
expect(setupContent).toContain('cleanup_prefixed_claude_symlinks');
|
||
const fnStart = setupContent.indexOf('cleanup_prefixed_claude_symlinks()');
|
||
const fnEnd = setupContent.indexOf('}', setupContent.indexOf('removed[@]}', fnStart));
|
||
const fnBody = setupContent.slice(fnStart, fnEnd);
|
||
expect(fnBody).toContain('readlink');
|
||
expect(fnBody).toContain('gstack-$skill_name');
|
||
});
|
||
|
||
test('reverse cleanup runs before link when prefix is disabled', () => {
|
||
const claudeInstallSection = setupContent.slice(
|
||
setupContent.indexOf('INSTALL_CLAUDE'),
|
||
setupContent.lastIndexOf('link_claude_skill_dirs')
|
||
);
|
||
expect(claudeInstallSection).toContain('cleanup_prefixed_claude_symlinks');
|
||
});
|
||
|
||
test('welcome message references SKILL_PREFIX', () => {
|
||
// gstack-upgrade is always called gstack-upgrade (it's the actual dir name)
|
||
// but the welcome section should exist near the prefix logic
|
||
expect(setupContent).toContain('Run /gstack-upgrade anytime');
|
||
});
|
||
});
|
||
|
||
describe('discover-skills hidden directory filtering', () => {
|
||
test('discoverTemplates skips dot-prefixed directories', () => {
|
||
const tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'gstack-discover-'));
|
||
try {
|
||
// Create a hidden dir with a template (should be excluded)
|
||
fs.mkdirSync(path.join(tmpDir, '.hidden'), { recursive: true });
|
||
fs.writeFileSync(path.join(tmpDir, '.hidden', 'SKILL.md.tmpl'), '---\nname: evil\n---\ntest');
|
||
// Create a visible dir with a template (should be included)
|
||
fs.mkdirSync(path.join(tmpDir, 'visible'), { recursive: true });
|
||
fs.writeFileSync(path.join(tmpDir, 'visible', 'SKILL.md.tmpl'), '---\nname: good\n---\ntest');
|
||
|
||
const { discoverTemplates } = require('../scripts/discover-skills');
|
||
const results = discoverTemplates(tmpDir);
|
||
const dirs = results.map((r: { tmpl: string }) => r.tmpl);
|
||
|
||
expect(dirs).toContain('visible/SKILL.md.tmpl');
|
||
expect(dirs).not.toContain('.hidden/SKILL.md.tmpl');
|
||
} finally {
|
||
fs.rmSync(tmpDir, { recursive: true, force: true });
|
||
}
|
||
});
|
||
});
|
||
|
||
describe('telemetry', () => {
|
||
test('telemetry start block lives in gstack-skill-start; render notes the handoff keys', () => {
|
||
// The start-block bash moved into the script (Phase 1): it reads the
|
||
// config, mints the session identity, and echoes the STATUS keys.
|
||
expect(SKILL_START_SCRIPT).toContain('_TEL_START=$(date +%s)');
|
||
expect(SKILL_START_SCRIPT).toContain('_SESSION_ID=');
|
||
expect(SKILL_START_SCRIPT).toContain('echo "TELEMETRY:');
|
||
expect(SKILL_START_SCRIPT).toContain('echo "TEL_PROMPTED:');
|
||
expect(SKILL_START_SCRIPT).toMatch(/gstack-config" get telemetry/);
|
||
// The render must tell the model to carry SESSION_ID/TEL_START to skill end.
|
||
const content = fs.readFileSync(path.join(ROOT, 'SKILL.md'), 'utf-8');
|
||
expect(content).toContain('SESSION_ID');
|
||
expect(content).toContain('TEL_START');
|
||
});
|
||
|
||
test('telemetry opt-in prompt lives in gstack-skill-start (marker-gated emit)', () => {
|
||
// Token-reduction Phase 2: the one-time consent prompt left the renders
|
||
// (absence pinned by test/onboarding-moved-literals.test.ts); the script
|
||
// gates it on the marker files and emits it as a GSTACK_INSTRUCTION block
|
||
// with all three config-set outcomes and the ack INSIDE the block.
|
||
expect(SKILL_START_SCRIPT).toContain(
|
||
'if [ "$_TEL_PROMPTED" = "no" ] && [ "$_LAKE_SEEN" = "yes" ]; then',
|
||
);
|
||
expect(SKILL_START_SCRIPT).toContain('_emit_block telemetry-prompt');
|
||
expect(SKILL_START_SCRIPT).toContain('gstack-config set telemetry community');
|
||
expect(SKILL_START_SCRIPT).toContain('gstack-config set telemetry anonymous');
|
||
expect(SKILL_START_SCRIPT).toContain('gstack-config set telemetry off');
|
||
expect(SKILL_START_SCRIPT).toContain('touch "$_GH/.telemetry-prompted"');
|
||
});
|
||
|
||
test('generated SKILL.md contains telemetry epilogue (one gstack-skill-end call)', () => {
|
||
const content = fs.readFileSync(path.join(ROOT, 'SKILL.md'), 'utf-8');
|
||
expect(content).toContain('Telemetry (run last)');
|
||
expect(content).toContain('gstack-skill-end --skill "gstack" --outcome OUTCOME');
|
||
expect(content).toContain('--tel-start "TEL_START"');
|
||
expect(content).toContain('PLAN MODE EXCEPTION');
|
||
// The duration math + remote-log dispatch moved into gstack-skill-end.
|
||
expect(SKILL_END_SCRIPT).toContain('_TEL_END');
|
||
expect(SKILL_END_SCRIPT).toContain('_TEL_DUR');
|
||
expect(SKILL_END_SCRIPT).toContain('SKILL_NAME');
|
||
expect(SKILL_END_SCRIPT).toContain('OUTCOME');
|
||
expect(SKILL_END_SCRIPT).toContain('gstack-telemetry-log');
|
||
});
|
||
|
||
test('pending marker handling lives in the scripts', () => {
|
||
// gstack-skill-start finalizes stale markers; gstack-skill-end clears the
|
||
// session's own marker.
|
||
expect(SKILL_START_SCRIPT).toContain("-name '.pending-*'");
|
||
expect(SKILL_START_SCRIPT).toContain('_pending_finalize');
|
||
expect(SKILL_END_SCRIPT).toContain('.pending-$SESSION_ID');
|
||
});
|
||
|
||
test('telemetry blocks appear in all skill files that use PREAMBLE', () => {
|
||
const skills = ['qa', 'ship', 'review', 'plan-ceo-review', 'plan-eng-review', 'retro'];
|
||
for (const skill of skills) {
|
||
const skillPath = path.join(ROOT, skill, 'SKILL.md');
|
||
if (fs.existsSync(skillPath)) {
|
||
const content = fs.readFileSync(skillPath, 'utf-8');
|
||
expect(content).toContain('Telemetry (run last)');
|
||
expect(content).toContain(`gstack-skill-end --skill "${skill}"`);
|
||
expect(content).toContain('--tel-start "TEL_START"');
|
||
}
|
||
}
|
||
});
|
||
});
|
||
|
||
describe('community fixes wave', () => {
|
||
// Helper to get all generated SKILL.md files
|
||
function getAllSkillMds(): Array<{ name: string; content: string }> {
|
||
const results: Array<{ name: string; content: string }> = [];
|
||
const rootPath = path.join(ROOT, 'SKILL.md');
|
||
if (fs.existsSync(rootPath)) {
|
||
results.push({ name: 'root', content: fs.readFileSync(rootPath, 'utf-8') });
|
||
}
|
||
for (const entry of fs.readdirSync(ROOT, { withFileTypes: true })) {
|
||
if (!entry.isDirectory() || entry.name.startsWith('.') || entry.name === 'node_modules') continue;
|
||
const skillPath = path.join(ROOT, entry.name, 'SKILL.md');
|
||
if (fs.existsSync(skillPath)) {
|
||
results.push({ name: entry.name, content: fs.readFileSync(skillPath, 'utf-8') });
|
||
}
|
||
}
|
||
return results;
|
||
}
|
||
|
||
// #594 — Discoverability: every SKILL.md.tmpl description contains "gstack"
|
||
test('every SKILL.md.tmpl description contains "gstack"', () => {
|
||
for (const skill of ALL_SKILLS) {
|
||
const tmplPath = skill.dir === '.' ? path.join(ROOT, 'SKILL.md.tmpl') : path.join(ROOT, skill.dir, 'SKILL.md.tmpl');
|
||
const content = fs.readFileSync(tmplPath, 'utf-8');
|
||
const desc = extractDescription(content);
|
||
expect(desc.toLowerCase()).toContain('gstack');
|
||
}
|
||
});
|
||
|
||
// #594 — Discoverability: first line of each description is under 120 chars
|
||
test('every SKILL.md.tmpl description first line is under 120 chars', () => {
|
||
for (const skill of ALL_SKILLS) {
|
||
const tmplPath = skill.dir === '.' ? path.join(ROOT, 'SKILL.md.tmpl') : path.join(ROOT, skill.dir, 'SKILL.md.tmpl');
|
||
const content = fs.readFileSync(tmplPath, 'utf-8');
|
||
const desc = extractDescription(content);
|
||
const firstLine = desc.split('\n')[0];
|
||
expect(firstLine.length).toBeLessThanOrEqual(120);
|
||
}
|
||
});
|
||
|
||
// #573 — Feature signals: ship/SKILL.md contains feature signal detection
|
||
test('ship/SKILL.md contains feature signal detection in Step 4', () => {
|
||
const content = readShipUnion();
|
||
expect(content.toLowerCase()).toContain('feature signal');
|
||
});
|
||
|
||
// #510 — Context warnings: no SKILL.md contains "running low on context"
|
||
test('no generated SKILL.md contains "running low on context"', () => {
|
||
const skills = getAllSkillMds();
|
||
for (const { name, content } of skills) {
|
||
expect(content).not.toContain('running low on context');
|
||
}
|
||
});
|
||
|
||
// #510 — Context warnings: plan-eng-review has explicit anti-warning
|
||
test('plan-eng-review/SKILL.md explicitly forbids preemptive context warnings', () => {
|
||
const content = readSkillUnion('plan-eng-review'); // carved: review body moved to section
|
||
expect(content.toLowerCase()).toContain('do not preemptively warn');
|
||
});
|
||
|
||
// #474 — Safety Net: no SKILL.md uses find with -delete
|
||
test('no generated SKILL.md contains find with -delete flag', () => {
|
||
const skills = getAllSkillMds();
|
||
for (const { name, content } of skills) {
|
||
// Match find commands that use -delete (but not prose mentioning the word "delete")
|
||
const lines = content.split('\n');
|
||
for (const line of lines) {
|
||
if (line.includes('find ') && line.includes('-delete')) {
|
||
throw new Error(`${name}/SKILL.md contains find with -delete: ${line.trim()}`);
|
||
}
|
||
}
|
||
}
|
||
});
|
||
|
||
// #467 — Telemetry: preamble JSONL writes are gated by telemetry setting
|
||
test('preamble JSONL writes are inside telemetry conditional', () => {
|
||
const preamble = fs.readFileSync(path.join(ROOT, 'scripts/resolvers/preamble.ts'), 'utf-8');
|
||
// Find all skill-usage.jsonl write lines
|
||
const lines = preamble.split('\n');
|
||
for (let i = 0; i < lines.length; i++) {
|
||
if (lines[i].includes('skill-usage.jsonl') && lines[i].includes('>>')) {
|
||
// Look backwards for a telemetry conditional within 5 lines
|
||
let foundConditional = false;
|
||
for (let j = i - 1; j >= Math.max(0, i - 5); j--) {
|
||
if (lines[j].includes('_TEL') && lines[j].includes('off')) {
|
||
foundConditional = true;
|
||
break;
|
||
}
|
||
}
|
||
expect(foundConditional).toBe(true);
|
||
}
|
||
}
|
||
});
|
||
});
|
||
|
||
describe('codex commands must not use inline $(git rev-parse --show-toplevel) for cwd', () => {
|
||
// Regression test: inline $(git rev-parse --show-toplevel) in codex exec -C
|
||
// or codex review without cd evaluates in whatever cwd the background shell
|
||
// inherits, which may be a different project in Conductor workspaces.
|
||
// The fix is to resolve _REPO_ROOT eagerly at the top of each bash block.
|
||
|
||
// Scan all source files that could contain codex commands
|
||
// Use Bun.Glob to avoid ELOOP from .claude/skills/gstack symlink back to ROOT
|
||
const tmplGlob = new Bun.Glob('**/*.tmpl');
|
||
const sourceFiles = [
|
||
...Array.from(tmplGlob.scanSync({ cwd: ROOT, followSymlinks: false })),
|
||
...fs.readdirSync(path.join(ROOT, 'scripts/resolvers'))
|
||
.filter(f => f.endsWith('.ts'))
|
||
.map(f => `scripts/resolvers/${f}`),
|
||
'scripts/gen-skill-docs.ts',
|
||
];
|
||
|
||
test('no codex exec command uses inline $(git rev-parse --show-toplevel) in -C flag', () => {
|
||
const violations: string[] = [];
|
||
for (const rel of sourceFiles) {
|
||
const abs = path.join(ROOT, rel);
|
||
if (!fs.existsSync(abs)) continue;
|
||
const content = fs.readFileSync(abs, 'utf-8');
|
||
const lines = content.split('\n');
|
||
for (let i = 0; i < lines.length; i++) {
|
||
const line = lines[i];
|
||
if (line.includes('codex exec') && line.includes('-C') && line.includes('$(git rev-parse --show-toplevel)')) {
|
||
violations.push(`${rel}:${i + 1}`);
|
||
}
|
||
}
|
||
}
|
||
expect(violations).toEqual([]);
|
||
});
|
||
|
||
test('no generated SKILL.md has codex exec with inline $(git rev-parse --show-toplevel) in -C flag', () => {
|
||
const violations: string[] = [];
|
||
const skillMdGlob = new Bun.Glob('**/SKILL.md');
|
||
const skillMdFiles = Array.from(skillMdGlob.scanSync({ cwd: ROOT, followSymlinks: false }));
|
||
for (const rel of skillMdFiles) {
|
||
const abs = path.join(ROOT, rel);
|
||
if (!fs.existsSync(abs)) continue;
|
||
const content = fs.readFileSync(abs, 'utf-8');
|
||
const lines = content.split('\n');
|
||
for (let i = 0; i < lines.length; i++) {
|
||
const line = lines[i];
|
||
if (line.includes('codex exec') && line.includes('-C') && line.includes('$(git rev-parse --show-toplevel)')) {
|
||
violations.push(`${rel}:${i + 1}`);
|
||
}
|
||
}
|
||
}
|
||
expect(violations).toEqual([]);
|
||
});
|
||
|
||
test('codex review commands must be preceded by cd "$_REPO_ROOT" (no -C support)', () => {
|
||
// codex review does not support -C, so the pattern must be:
|
||
// _REPO_ROOT=$(git rev-parse --show-toplevel) || { ... }
|
||
// cd "$_REPO_ROOT"
|
||
// codex review ...
|
||
// NOT: codex review ... with inline $(git rev-parse --show-toplevel)
|
||
const allFiles = [
|
||
...Array.from(tmplGlob.scanSync({ cwd: ROOT, followSymlinks: false })),
|
||
...Array.from(new Bun.Glob('**/SKILL.md').scanSync({ cwd: ROOT, followSymlinks: false })),
|
||
...fs.readdirSync(path.join(ROOT, 'scripts/resolvers'))
|
||
.filter(f => f.endsWith('.ts'))
|
||
.map(f => `scripts/resolvers/${f}`),
|
||
'scripts/gen-skill-docs.ts',
|
||
];
|
||
const violations: string[] = [];
|
||
for (const rel of allFiles) {
|
||
const abs = path.join(ROOT, rel);
|
||
if (!fs.existsSync(abs)) continue;
|
||
const content = fs.readFileSync(abs, 'utf-8');
|
||
const lines = content.split('\n');
|
||
for (let i = 0; i < lines.length; i++) {
|
||
const line = lines[i];
|
||
// Skip non-executable lines (markdown table cells, prose references)
|
||
if (line.includes('|') && line.includes('`/codex review`')) continue;
|
||
if (line.includes('`codex review`')) continue;
|
||
// Check for codex review with inline $(git rev-parse)
|
||
if (line.includes('codex review') && line.includes('$(git rev-parse --show-toplevel)')) {
|
||
violations.push(`${rel}:${i + 1} — inline git rev-parse in codex review`);
|
||
}
|
||
}
|
||
}
|
||
expect(violations).toEqual([]);
|
||
});
|
||
|
||
test('codex review commands take their scope from a flag, never from prompt text', () => {
|
||
// `codex review` scope comes ONLY from --base/--commit/--uncommitted. The
|
||
// positional [PROMPT] is mutually exclusive with all three (#1428, #1479),
|
||
// and a prompt-only `codex review` silently falls back to the *uncommitted
|
||
// working-tree* scope (`git status --short; git diff`) — so describing the
|
||
// diff range in prompt text produces a confident review of the wrong
|
||
// changes, with no error. Both halves are pinned here:
|
||
// (a) every `codex review` invocation carries a scope flag, and
|
||
// (b) no invocation puts a positional prompt in front of that flag.
|
||
//
|
||
// This does NOT apply to `codex exec`, which is agentic and really does run
|
||
// the git command it's told to — the adversarial pass legitimately scopes
|
||
// itself in prompt text.
|
||
const checkedFiles = [
|
||
'scripts/resolvers/review.ts',
|
||
'review/SKILL.md',
|
||
'ship/SKILL.md',
|
||
'codex/SKILL.md.tmpl',
|
||
'codex/SKILL.md',
|
||
// codex's scoped invocations moved into the carved review-mode section
|
||
// (T9) — keep sweeping both the .tmpl source and the generated section.
|
||
'codex/sections/review-mode.md.tmpl',
|
||
'codex/sections/review-mode.md',
|
||
];
|
||
|
||
const violations: string[] = [];
|
||
for (const rel of checkedFiles) {
|
||
// ship's codex/adversarial command moved into sections/adversarial.md (T9 carve).
|
||
const content = rel === 'ship/SKILL.md' ? readShipUnion() : fs.readFileSync(path.join(ROOT, rel), 'utf-8');
|
||
const lines = content.split('\n');
|
||
for (let i = 0; i < lines.length; i++) {
|
||
const line = lines[i];
|
||
// Only inspect real shell invocations, not prose mentioning the command.
|
||
if (line.includes('`codex review`')) continue;
|
||
const match = line.match(/(?:^|[;&|]\s*|\s)codex\s+review\b(.*)$/);
|
||
if (!match) continue;
|
||
const rest = match[1];
|
||
const scopeFlag = /--base\b|--commit\b|--uncommitted\b/;
|
||
if (!scopeFlag.test(rest)) {
|
||
// A quoted prompt with no scope flag is the silent-wrong-scope bug.
|
||
if (/^\s*["'$]/.test(rest)) {
|
||
violations.push(`${rel}:${i + 1} — prompt-only codex review (falls back to working-tree scope)`);
|
||
}
|
||
continue;
|
||
}
|
||
const beforeFlag = rest.split(scopeFlag)[0].trim();
|
||
if (/^["'$]|^--\s*["']/.test(beforeFlag)) {
|
||
violations.push(`${rel}:${i + 1} — positional prompt passed alongside a scope flag`);
|
||
}
|
||
}
|
||
}
|
||
expect(violations).toEqual([]);
|
||
});
|
||
});
|
||
|
||
// ─── Learnings + Confidence Resolver Tests ─────────────────────
|
||
|
||
describe('LEARNINGS_SEARCH resolver', () => {
|
||
const SEARCH_SKILLS = ['review', 'ship', 'plan-eng-review', 'investigate', 'office-hours', 'plan-ceo-review'];
|
||
|
||
for (const skill of SEARCH_SKILLS) {
|
||
test(`${skill} generated SKILL.md contains learnings search`, () => {
|
||
const content = readSkillUnion(skill); // ship: moved to sections/plan-completion.md
|
||
expect(content).toContain('Prior Learnings');
|
||
expect(content).toContain('gstack-learnings-search');
|
||
});
|
||
}
|
||
|
||
test('learnings search includes cross-project config check', () => {
|
||
const content = fs.readFileSync(path.join(ROOT, 'review', 'SKILL.md'), 'utf-8');
|
||
expect(content).toContain('cross_project_learnings');
|
||
expect(content).toContain('--cross-project');
|
||
});
|
||
|
||
test('learnings search includes AskUserQuestion for first-time cross-project opt-in', () => {
|
||
const content = fs.readFileSync(path.join(ROOT, 'review', 'SKILL.md'), 'utf-8');
|
||
expect(content).toContain('Enable cross-project learnings');
|
||
expect(content).toContain('project-scoped only');
|
||
});
|
||
|
||
test('Eng clarifies its full brief without changing any host options or other skills', async () => {
|
||
const { generateLearningsSearch } = await import('../scripts/resolvers/learnings');
|
||
const { ALL_HOST_CONFIGS } = await import('../hosts');
|
||
const { HOST_PATHS } = await import('../scripts/resolvers/types');
|
||
const fullBrief = 'Build a full decision brief from these facts and options using the preamble format, then ask and wait:';
|
||
for (const host of ALL_HOST_CONFIGS) {
|
||
const ctx = { skillName: 'plan-eng-review', tmplPath: 'plan-eng-review/SKILL.md.tmpl', host: host.name, paths: HOST_PATHS[host.name]! };
|
||
const eng = generateLearningsSearch(ctx);
|
||
const standard = generateLearningsSearch({ ...ctx, skillName: 'review' });
|
||
expect(generateLearningsSearch({ ...ctx, skillName: 'plan-ceo-review' })).toBe(standard);
|
||
if (host.learningsMode === 'basic') {
|
||
expect(eng).toBe(standard);
|
||
expect(eng).not.toContain('CROSS_PROJECT');
|
||
} else {
|
||
expect(eng).toContain(fullBrief);
|
||
expect(eng.replace(fullBrief, 'Use AskUserQuestion:')).toBe(standard);
|
||
expect(eng).toContain('A) Enable cross-project learnings (recommended)');
|
||
expect(eng).toContain('B) Keep learnings project-scoped only');
|
||
expect(eng).toContain('gstack-config set cross_project_learnings true');
|
||
expect(eng).toContain('gstack-config set cross_project_learnings false');
|
||
}
|
||
}
|
||
});
|
||
|
||
test('learnings search mentions prior learning applied display format', () => {
|
||
const content = fs.readFileSync(path.join(ROOT, 'review', 'SKILL.md'), 'utf-8');
|
||
expect(content).toContain('Prior learning applied');
|
||
});
|
||
});
|
||
|
||
describe('LEARNINGS_LOG resolver', () => {
|
||
const LOG_SKILLS = ['review', 'retro', 'investigate'];
|
||
|
||
for (const skill of LOG_SKILLS) {
|
||
test(`${skill} generated SKILL.md contains learnings log`, () => {
|
||
const content = fs.readFileSync(path.join(ROOT, skill, 'SKILL.md'), 'utf-8');
|
||
expect(content).toContain('Capture Learnings');
|
||
expect(content).toContain('gstack-learnings-log');
|
||
});
|
||
}
|
||
|
||
test('learnings log documents all type values', () => {
|
||
const content = fs.readFileSync(path.join(ROOT, 'review', 'SKILL.md'), 'utf-8');
|
||
for (const type of ['pattern', 'pitfall', 'preference', 'architecture', 'tool']) {
|
||
expect(content).toContain(type);
|
||
}
|
||
});
|
||
|
||
test('learnings log documents all source values', () => {
|
||
const content = fs.readFileSync(path.join(ROOT, 'review', 'SKILL.md'), 'utf-8');
|
||
for (const source of ['observed', 'user-stated', 'inferred', 'cross-model']) {
|
||
expect(content).toContain(source);
|
||
}
|
||
});
|
||
|
||
test('learnings log includes files field for staleness detection', () => {
|
||
const content = fs.readFileSync(path.join(ROOT, 'review', 'SKILL.md'), 'utf-8');
|
||
expect(content).toContain('"files"');
|
||
expect(content).toContain('staleness detection');
|
||
});
|
||
});
|
||
|
||
describe('CONFIDENCE_CALIBRATION resolver', () => {
|
||
// CSO owns a distinct evidence rubric; the shared numerical confidence
|
||
// resolver would conflict with that contract and its private startup.
|
||
const CONFIDENCE_SKILLS = ['review', 'ship', 'plan-eng-review'];
|
||
|
||
for (const skill of CONFIDENCE_SKILLS) {
|
||
test(`${skill} generated SKILL.md contains confidence calibration`, () => {
|
||
const content = readSkillUnion(skill); // ship: moved to sections/review-army.md
|
||
expect(content).toContain('Confidence Calibration');
|
||
expect(content).toContain(skill === 'review' ? 'score every finding (1-10)' : 'confidence score');
|
||
});
|
||
}
|
||
|
||
test('confidence calibration includes scoring rubric with all tiers', () => {
|
||
const content = fs.readFileSync(path.join(ROOT, 'review', 'SKILL.md'), 'utf-8');
|
||
expect(content).toContain('9-10');
|
||
expect(content).toContain('7-8');
|
||
expect(content).toContain('5-6');
|
||
expect(content).toContain('3-4');
|
||
expect(content).toContain('1-2');
|
||
});
|
||
|
||
test('confidence calibration includes display rules', () => {
|
||
const content = fs.readFileSync(path.join(ROOT, 'review', 'SKILL.md'), 'utf-8');
|
||
expect(content).toContain('Show normally');
|
||
expect(content).toContain('Suppress from main report');
|
||
});
|
||
|
||
test('confidence calibration includes finding format example', () => {
|
||
const content = fs.readFileSync(path.join(ROOT, 'review', 'SKILL.md'), 'utf-8');
|
||
expect(content).toContain('[CRITICAL] (confidence:');
|
||
expect(content).toContain('SQL injection');
|
||
});
|
||
|
||
test('confidence calibration includes calibration learning feedback loop', () => {
|
||
const content = fs.readFileSync(path.join(ROOT, 'review', 'SKILL.md'), 'utf-8');
|
||
const flat = content.replace(/\s+/g, ' ');
|
||
expect(flat).toContain('Calibration learning');
|
||
expect(flat).toContain('If the user confirms a reported finding scored < 7 is real');
|
||
expect(flat).toContain('log the corrected pattern as a learning');
|
||
});
|
||
|
||
test('skills without confidence calibration do NOT contain it', () => {
|
||
// office-hours and retro do NOT use confidence calibration
|
||
for (const skill of ['office-hours', 'retro']) {
|
||
const content = fs.readFileSync(path.join(ROOT, skill, 'SKILL.md'), 'utf-8');
|
||
expect(content).not.toContain('## Confidence Calibration');
|
||
}
|
||
});
|
||
});
|
||
|
||
describe('gen-skill-docs prefix warning (#620/#578)', () => {
|
||
const { execSync } = require('child_process');
|
||
|
||
test('warns about skill_prefix when config has prefix=true', () => {
|
||
const tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'gstack-prefix-warn-'));
|
||
try {
|
||
// Create a fake ~/.gstack/config.yaml with skill_prefix: true
|
||
const fakeHome = tmpDir;
|
||
const fakeGstack = path.join(fakeHome, '.gstack');
|
||
fs.mkdirSync(fakeGstack, { recursive: true });
|
||
fs.writeFileSync(path.join(fakeGstack, 'config.yaml'), 'skill_prefix: true\n');
|
||
|
||
// Render into an out-dir under the fixture (the warning fires on any
|
||
// non-dry-run generation) so the live tree is never rewritten.
|
||
const outDir = path.join(tmpDir, 'out');
|
||
const output = execSync(`bun run scripts/gen-skill-docs.ts --out-dir "${outDir}"`, {
|
||
cwd: ROOT,
|
||
env: { ...process.env, HOME: fakeHome },
|
||
encoding: 'utf-8',
|
||
timeout: 30000,
|
||
});
|
||
expect(output).toContain('skill_prefix is true');
|
||
expect(output).toContain('gstack-relink');
|
||
} finally {
|
||
fs.rmSync(tmpDir, { recursive: true, force: true });
|
||
}
|
||
});
|
||
|
||
test('no warning when skill_prefix is false or absent', () => {
|
||
const tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'gstack-prefix-warn-'));
|
||
try {
|
||
const fakeHome = tmpDir;
|
||
const fakeGstack = path.join(fakeHome, '.gstack');
|
||
fs.mkdirSync(fakeGstack, { recursive: true });
|
||
fs.writeFileSync(path.join(fakeGstack, 'config.yaml'), 'skill_prefix: false\n');
|
||
|
||
const outDir = path.join(tmpDir, 'out');
|
||
const output = execSync(`bun run scripts/gen-skill-docs.ts --out-dir "${outDir}"`, {
|
||
cwd: ROOT,
|
||
env: { ...process.env, HOME: fakeHome },
|
||
encoding: 'utf-8',
|
||
timeout: 30000,
|
||
});
|
||
expect(output).not.toContain('skill_prefix is true');
|
||
} finally {
|
||
fs.rmSync(tmpDir, { recursive: true, force: true });
|
||
}
|
||
});
|
||
});
|
||
|
||
describe('voice-triggers processing', () => {
|
||
const { extractVoiceTriggers, processVoiceTriggers } = require('../scripts/gen-skill-docs') as {
|
||
extractVoiceTriggers: (content: string) => string[];
|
||
processVoiceTriggers: (content: string) => string;
|
||
};
|
||
|
||
test('extractVoiceTriggers parses valid YAML list', () => {
|
||
const content = `---\nname: cso\ndescription: |\n Security audit.\nvoice-triggers:\n - "see-so"\n - "security review"\n---\nBody`;
|
||
const triggers = extractVoiceTriggers(content);
|
||
expect(triggers).toEqual(['see-so', 'security review']);
|
||
});
|
||
|
||
test('extractVoiceTriggers returns [] when no field present', () => {
|
||
const content = `---\nname: qa\ndescription: |\n QA testing.\n---\nBody`;
|
||
expect(extractVoiceTriggers(content)).toEqual([]);
|
||
});
|
||
|
||
test('processVoiceTriggers appends voice triggers to description', () => {
|
||
const content = `---\nname: cso\ndescription: |\n Security audit. (gstack)\nvoice-triggers:\n - "see-so"\n - "security review"\n---\nBody`;
|
||
const result = processVoiceTriggers(content);
|
||
expect(result).toContain('Voice triggers (speech-to-text aliases): "see-so", "security review".');
|
||
});
|
||
|
||
test('processVoiceTriggers strips voice-triggers field from output', () => {
|
||
const content = `---\nname: cso\ndescription: |\n Security audit. (gstack)\nvoice-triggers:\n - "see-so"\n---\nBody`;
|
||
const result = processVoiceTriggers(content);
|
||
expect(result).not.toContain('voice-triggers:');
|
||
});
|
||
|
||
test('processVoiceTriggers returns content unchanged when no voice-triggers', () => {
|
||
const content = `---\nname: qa\ndescription: |\n QA testing.\n---\nBody`;
|
||
expect(processVoiceTriggers(content)).toBe(content);
|
||
});
|
||
|
||
test('generated CSO SKILL.md contains voice triggers in description', () => {
|
||
const content = fs.readFileSync(path.join(ROOT, 'cso', 'SKILL.md'), 'utf-8');
|
||
expect(content).toContain('"see-so"');
|
||
expect(content).toContain('Voice triggers (speech-to-text aliases):');
|
||
});
|
||
|
||
test('generated CSO SKILL.md does NOT contain raw voice-triggers field', () => {
|
||
const content = fs.readFileSync(path.join(ROOT, 'cso', 'SKILL.md'), 'utf-8');
|
||
const fmEnd = content.indexOf('\n---', 4);
|
||
const frontmatter = content.slice(0, fmEnd);
|
||
expect(frontmatter).not.toContain('voice-triggers:');
|
||
});
|
||
|
||
test('generated Claude CSO skill preauthorizes only the trusted launcher', () => {
|
||
const expected = [
|
||
'Bash(~/.claude/skills/gstack/bin/gstack-cso-launcher *)',
|
||
'Bash(~/.claude/skills/gstack/bin/gstack-cso-launcher.exe *)',
|
||
];
|
||
for (const file of ['cso/SKILL.md.tmpl', 'cso/SKILL.md']) {
|
||
const content = fs.readFileSync(path.join(ROOT, file), 'utf-8');
|
||
const fmEnd = content.indexOf('\n---', 4);
|
||
const frontmatter = Bun.YAML.parse(content.slice(4, fmEnd)) as Record<string, unknown>;
|
||
expect(frontmatter['allowed-tools'], file).toEqual(expected);
|
||
}
|
||
});
|
||
|
||
test('generated CSO host variants retain challenge fallback and host-containment disclosure', () => {
|
||
const claude = fs.readFileSync(path.join(ROOT, 'cso', 'SKILL.md'), 'utf-8');
|
||
const codex = fs.readFileSync(path.join(EXTERNAL_OUT, '.agents', 'skills', 'gstack-cso', 'SKILL.md'), 'utf-8');
|
||
for (const content of [claude, codex]) {
|
||
expect(content).toContain('sequential challenge; independent agent unavailable');
|
||
expect(content).toContain('Containment does not sandbox the host agent or kernel.');
|
||
expect(content).toContain('Do not request broader tool access solely to obtain an independent reviewer.');
|
||
}
|
||
});
|
||
|
||
// Gen-time-only keys: interactive + benefits-from are read from the .tmpl by
|
||
// buildContext; the generated copy has no reader (the host reads name/
|
||
// description/allowed-tools/hooks; gbrain: is runtime-read and NOT stripped).
|
||
// Pin the strip so a stripFields refactor can't silently re-add the always-on
|
||
// frontmatter weight — mirrors the voice-triggers pins above.
|
||
test('generated SKILL.md strips gen-time-only keys the .tmpl still declares', () => {
|
||
const tmpl = fs.readFileSync(path.join(ROOT, 'plan-ceo-review', 'SKILL.md.tmpl'), 'utf-8');
|
||
const tmplFm = tmpl.slice(0, tmpl.indexOf('\n---', 4));
|
||
expect(tmplFm).toContain('interactive:');
|
||
expect(tmplFm).toContain('benefits-from:');
|
||
|
||
const generated = fs.readFileSync(path.join(ROOT, 'plan-ceo-review', 'SKILL.md'), 'utf-8');
|
||
const genFm = generated.slice(0, generated.indexOf('\n---', 4));
|
||
expect(genFm).not.toContain('interactive:');
|
||
expect(genFm).not.toContain('benefits-from:');
|
||
|
||
// The runtime-read and host-read keys survive the strip.
|
||
const investigate = fs.readFileSync(path.join(ROOT, 'investigate', 'SKILL.md'), 'utf-8');
|
||
const invFm = investigate.slice(0, investigate.indexOf('\n---', 4));
|
||
expect(invFm).toContain('hooks:');
|
||
expect(invFm).toContain('gbrain:');
|
||
});
|
||
});
|
||
|
||
describe('CEO accepted requirement preservation', () => {
|
||
test('HOLD SCOPE includes implementation work needed to meet accepted requirements', () => {
|
||
const main = fs.readFileSync(path.join(ROOT, 'plan-ceo-review/SKILL.md'), 'utf8');
|
||
const hold = main.slice(main.indexOf('**For HOLD SCOPE**'), main.indexOf('**For SCOPE REDUCTION**'));
|
||
expect(hold).toContain('Keep stated invariants and acceptance criteria; repairs needed to meet them are in scope.');
|
||
});
|
||
|
||
test('loaded sections keep requirement conflicts unresolved until an authorized decision', () => {
|
||
const section = fs.readFileSync(path.join(ROOT, 'plan-ceo-review/sections/review-sections.md'), 'utf8');
|
||
const policy = section.slice(section.indexOf('**Preserve accepted requirements.**'), section.indexOf('### Section 1:')).replace(/\s+/g, ' ');
|
||
expect(policy).toContain('Report gaps and propose remedies, including omitted mechanisms in HOLD SCOPE');
|
||
expect(policy).toContain('Never weaken a guarantee, accept its violation or change a test to expect it');
|
||
expect(policy).toContain('Low frequency, bounded impact and documentation do not meet stricter requirements');
|
||
expect(policy).toContain('keep both the proposal and original gap unresolved');
|
||
expect(policy).toContain('Changing a requirement needs explicit authority');
|
||
expect(policy).toContain('Routine auto-decide cannot override user constraints or non-goals');
|
||
expect(policy).toContain('Carry prior approvals into findings, tasks and the report');
|
||
});
|
||
});
|
||
|
||
describe('plan-mode-info resolver (handshake-replacement)', () => {
|
||
const REVIEW_SKILLS = [
|
||
'plan-ceo-review',
|
||
'plan-eng-review',
|
||
'plan-design-review',
|
||
'plan-devex-review',
|
||
];
|
||
|
||
// Header for the vestigial handshake that was removed. If it ever reappears,
|
||
// someone accidentally re-introduced the resolver.
|
||
const HANDSHAKE_MARKER = '## Plan Mode Handshake';
|
||
// Header for the new plan-mode-info section (previously lived at the tail
|
||
// of completion-status.ts; now hoisted to position 1 of the preamble).
|
||
const PLAN_MODE_INFO_MARKER = '## Skill Invocation During Plan Mode';
|
||
|
||
test('vestigial handshake is absent from all generated Claude SKILL.md files', () => {
|
||
// Scan every generated SKILL.md under ROOT (top-level directory per skill).
|
||
// Using fs.readdirSync + filter instead of a glob so we catch any skill
|
||
// that gets added later without updating this list.
|
||
const entries = fs.readdirSync(ROOT, { withFileTypes: true });
|
||
let checked = 0;
|
||
for (const entry of entries) {
|
||
if (!entry.isDirectory()) continue;
|
||
const skillMd = path.join(ROOT, entry.name, 'SKILL.md');
|
||
if (!fs.existsSync(skillMd)) continue;
|
||
const content = fs.readFileSync(skillMd, 'utf-8');
|
||
expect(content, `handshake marker in ${entry.name}/SKILL.md`).not.toContain(HANDSHAKE_MARKER);
|
||
checked++;
|
||
}
|
||
expect(checked).toBeGreaterThan(0);
|
||
});
|
||
|
||
test('vestigial handshake is absent from non-Claude host outputs', () => {
|
||
// Non-Claude hosts render to hostSubdirs (.agents/, .openclaw/, etc). The
|
||
// plan-mode-info resolver has no host-scoping — all hosts get the new
|
||
// section, none get the old handshake. Scan every candidate host tree in
|
||
// the module-level out-dir render (--host all), which is always present —
|
||
// so the check can no longer silently degrade to a console warning.
|
||
const hostDirs = ['.agents', '.openclaw', '.opencode', '.factory', '.hermes', '.kiro', '.cursor', '.slate'];
|
||
let checked = 0;
|
||
for (const host of hostDirs) {
|
||
const skillsRoot = path.join(EXTERNAL_OUT, host, 'skills');
|
||
if (!fs.existsSync(skillsRoot)) continue;
|
||
const entries = fs.readdirSync(skillsRoot, { withFileTypes: true });
|
||
for (const entry of entries) {
|
||
if (!entry.isDirectory()) continue;
|
||
const skillMd = path.join(skillsRoot, entry.name, 'SKILL.md');
|
||
if (!fs.existsSync(skillMd)) continue;
|
||
const content = fs.readFileSync(skillMd, 'utf-8');
|
||
expect(content, `handshake marker in ${host}/skills/${entry.name}/SKILL.md`).not.toContain(HANDSHAKE_MARKER);
|
||
checked++;
|
||
}
|
||
}
|
||
expect(checked).toBeGreaterThan(0);
|
||
});
|
||
|
||
test.each(REVIEW_SKILLS)(
|
||
'%s/SKILL.md contains the new plan-mode-info section near the top',
|
||
(skill) => {
|
||
const content = fs.readFileSync(path.join(ROOT, skill, 'SKILL.md'), 'utf-8');
|
||
const idx = content.indexOf(PLAN_MODE_INFO_MARKER);
|
||
expect(idx).toBeGreaterThan(0);
|
||
// Position 1 in preamble composition = within the first ~300 lines.
|
||
// Roughly translates to first ~15KB of text.
|
||
expect(idx).toBeLessThan(15_000);
|
||
},
|
||
);
|
||
|
||
test('plan-mode-info is wired BEFORE generateUpgradeCheck in preamble', () => {
|
||
// Token-reduction Phase 2: generateUpgradeCheck's render output is now
|
||
// ONLY the steady-state PROACTIVE-false + SKILL_PREFIX rules (the
|
||
// UPGRADE_AVAILABLE prose emits from bin/gstack-skill-start at runtime),
|
||
// so those rules are the resolver's order marker.
|
||
const content = fs.readFileSync(
|
||
path.join(ROOT, 'plan-ceo-review', 'SKILL.md'),
|
||
'utf-8',
|
||
);
|
||
const planModeIdx = content.indexOf(PLAN_MODE_INFO_MARKER);
|
||
const upgradeIdx = content.indexOf('If `PROACTIVE` is `"false"`');
|
||
expect(planModeIdx).toBeGreaterThan(0);
|
||
expect(upgradeIdx).toBeGreaterThan(0);
|
||
expect(planModeIdx).toBeLessThan(upgradeIdx);
|
||
});
|
||
|
||
test('0D authority and fresh-approval paths precede mode selection', () => {
|
||
const content = fs.readFileSync(path.join(ROOT, 'plan-ceo-review', 'SKILL.md'), 'utf-8').replace(/\s+/g, ' ');
|
||
const approachIdx = content.indexOf('### 0D.');
|
||
const constructIdx = content.indexOf('Build one `currentDecision`', approachIdx);
|
||
const saveIdx = content.indexOf('**Pre-question checkpoint:**', constructIdx);
|
||
const presentIdx = content.indexOf('Ask one row per call', saveIdx);
|
||
const stopIdx = content.indexOf('**STOP for the actual answer, even for a lone option.**', presentIdx);
|
||
const modeIdx = content.indexOf('### 0E. Mode Selection');
|
||
const preludeIdx = content.indexOf('### 0F');
|
||
const positions = [approachIdx, constructIdx, saveIdx, presentIdx, stopIdx, modeIdx, preludeIdx];
|
||
expect(positions.every(position => position > 0)).toBe(true);
|
||
expect(positions).toEqual([...positions].sort((a, b) => a - b));
|
||
const approach = content.slice(approachIdx, modeIdx);
|
||
expect(approach).toMatch(/Compare input, source and answers/);
|
||
expect(approach).toMatch(/correct facts, flag conflicts and preserve unknowns/i);
|
||
expect(approach).toMatch(/Reuse (?:an )?exact approvals?/);
|
||
expect(content).toMatch(/(?:the actual instruction\/answer|Cite actual instructions\/answers) and exact scope/);
|
||
const reopenRule = approach.match(/Reuse exact approvals\. Reopen only for contradictions, changed assumptions or user instructions, never speculation or reviewer agreement/)?.[0];
|
||
expect(reopenRule).toBeDefined();
|
||
expect(reopenRule!).toMatch(/reopen only|requires reopening it/i);
|
||
expect(content.indexOf(reopenRule!)).toBeGreaterThan(approachIdx);
|
||
expect(content.indexOf(reopenRule!)).toBeLessThan(presentIdx);
|
||
const gate = content.slice(stopIdx, modeIdx);
|
||
const startup = content.slice(content.indexOf('### 0C.'), approachIdx);
|
||
expect(startup).toContain('Before 0E, call 0D for unresolved approaches');
|
||
expect(startup).toContain('A) current/requested plan, B) smallest scoped alternative');
|
||
expect(startup).toContain('With no required choice, or after those choices settle, go to 0E');
|
||
expect(approach).toContain('0D returns to its caller, not to mode selection');
|
||
expect(approach).toContain("For mode changes, follow 0E's **Mode change** instruction");
|
||
const modeChange = content.slice(content.indexOf('**Mode change:**'), preludeIdx);
|
||
expect(modeChange).toContain('Pause and ask with the four-mode menu; keep the mode until answered');
|
||
expect(modeChange).toContain('complete newly applicable Step 0 work in route order, reusing completed work and scope answers');
|
||
expect(modeChange).toContain('Then resume the paused step. If unchanged, resume directly');
|
||
expect(gate).toContain('Return to the calling step with the saved answer; do not ask it again');
|
||
expect(gate).not.toContain("When this step's required decisions are settled, go to 0E if you came from 0C");
|
||
expect(gate).toContain('even for a lone option');
|
||
expect(approach).toContain('A recommendation is not approval');
|
||
expect(approach).toMatch(/Ask one row per call with that object unchanged, without recomposing/);
|
||
expect(approach).toContain('Compare its question, header, labels and full descriptions literally with the verified fields');
|
||
expect(gate).toMatch(/(?:Record its reference|Save the answer reference) and scope in Exact approval and scope/);
|
||
expect(gate).toMatch(/update Status,? and amend only (?:what it authorizes|authorized work)/);
|
||
expect(gate).toContain('storage policy before taking another row');
|
||
expect(approach).not.toContain('Do NOT proceed to Step 0D or 0F until the user responds to 0C-bis');
|
||
});
|
||
});
|
||
|
||
// GSTACK REVIEW REPORT report-at-bottom contract — verifies the prompt-text
|
||
// fix in scripts/resolvers/review.ts (the load-bearing change for the
|
||
// "report not at bottom of plan in plan mode" bug). The bug is in the
|
||
// prompt's contradictory write-flow instructions, not in observable
|
||
// runtime behavior we can cheaply gate in CI. Verifying the prompt text
|
||
// directly is the deterministic equivalent of the regression test the
|
||
// PTY harness can't reliably drive (autoplan needs auto-progression of
|
||
// AskUserQuestions to reach the report-write step, which the harness
|
||
// doesn't support today).
|
||
describe('GSTACK REVIEW REPORT delete-then-append flow', () => {
|
||
const PLAN_REVIEW_SKILLS = [
|
||
'plan-ceo-review',
|
||
'plan-design-review',
|
||
'plan-devex-review',
|
||
'plan-eng-review',
|
||
];
|
||
|
||
for (const skill of PLAN_REVIEW_SKILLS) {
|
||
test(`${skill}/SKILL.md prescribes delete-then-append, not in-place replace`, () => {
|
||
// Carved skills (v2 plan Phase B) relocate the review-report prose into
|
||
// sections/*.md; readSkillUnion follows the content wherever the carve put it.
|
||
const content = readSkillUnion(skill);
|
||
|
||
// The new (correct) instruction must be present.
|
||
expect(content).toContain('delete-then-append flow');
|
||
expect(content).toContain('never mid-file');
|
||
expect(content).toContain('Do NOT replace the section in place');
|
||
|
||
// The old contradictory bullets must be gone. The signature phrase
|
||
// from the buggy prompt was 'replace it entirely using the Edit tool'
|
||
// which is what allowed mid-file reports to stay mid-file.
|
||
expect(content).not.toContain('replace it** entirely using the Edit tool');
|
||
expect(content).not.toContain('If it was found mid-file, move it');
|
||
});
|
||
}
|
||
|
||
test('scripts/resolvers/review.ts source has the rewritten flow', () => {
|
||
const src = fs.readFileSync(path.join(ROOT, 'scripts', 'resolvers', 'review.ts'), 'utf-8');
|
||
expect(src).toContain('delete-then-append flow');
|
||
expect(src).toContain('never mid-file');
|
||
expect(src).toContain('Do NOT replace the section in place');
|
||
// Old contradictory bullets are gone from the source resolver.
|
||
expect(src).not.toContain('replace it** entirely using the Edit tool');
|
||
expect(src).not.toContain('If it was found mid-file, move it');
|
||
});
|
||
});
|
||
|
||
describe('LEARNINGS_SEARCH resolver: query parameter', () => {
|
||
// Lazy-load resolver and types after describe block to keep test file self-contained.
|
||
const { generateLearningsSearch } = require('../scripts/resolvers/learnings');
|
||
const { HOST_PATHS } = require('../scripts/resolvers/types');
|
||
|
||
const claudeCtx = {
|
||
skillName: 'test',
|
||
tmplPath: 'test/SKILL.md.tmpl',
|
||
host: 'claude',
|
||
paths: HOST_PATHS.claude,
|
||
};
|
||
const codexCtx = { ...claudeCtx, host: 'codex', paths: HOST_PATHS.codex };
|
||
|
||
test('no args → bash does not contain --query (backwards-compat)', () => {
|
||
const out = generateLearningsSearch(claudeCtx);
|
||
expect(out).not.toContain('--query');
|
||
});
|
||
|
||
test('claude host + query=foo bar → both cross-project and project-scoped branches contain --query', () => {
|
||
const out = generateLearningsSearch(claudeCtx, ['query=foo bar']);
|
||
// Both branches of the if/else must carry the flag.
|
||
const lines = out.split('\n').filter(l => l.includes('gstack-learnings-search'));
|
||
expect(lines.length).toBeGreaterThanOrEqual(2);
|
||
for (const line of lines) {
|
||
expect(line).toContain('--query "foo bar"');
|
||
}
|
||
});
|
||
|
||
test('codex host + query=foo bar → codex bash variant contains --query', () => {
|
||
const out = generateLearningsSearch(codexCtx, ['query=foo bar']);
|
||
expect(out).toContain('--query "foo bar"');
|
||
expect(out).toContain('$GSTACK_BIN/gstack-learnings-search');
|
||
});
|
||
|
||
test('empty value query= → bash does not contain --query (locked semantics: falls through)', () => {
|
||
const claudeOut = generateLearningsSearch(claudeCtx, ['query=']);
|
||
expect(claudeOut).not.toContain('--query');
|
||
const codexOut = generateLearningsSearch(codexCtx, ['query=']);
|
||
expect(codexOut).not.toContain('--query');
|
||
});
|
||
|
||
test('shell-injection chars in query= → throws at gen-time (defense in depth)', () => {
|
||
for (const bad of ['$(whoami)', '`cmd`', 'a;b', 'a&b', 'a"b', 'a\\b', 'foo$x']) {
|
||
expect(() => generateLearningsSearch(claudeCtx, [`query=${bad}`])).toThrow(/alphanumeric/);
|
||
}
|
||
});
|
||
});
|
||
|
||
describe('EXIT PLAN MODE GATE placement', () => {
|
||
// Fresh skill list — do NOT reuse REVIEW_SKILLS upstream (3 entries, missing plan-devex).
|
||
const planSkills = ['plan-eng-review', 'plan-ceo-review', 'plan-design-review', 'plan-devex-review'];
|
||
|
||
// Strip fenced code blocks before matching headings — PLAN_FILE_REVIEW_REPORT
|
||
// already contains `## GSTACK REVIEW REPORT` inside a markdown example fence,
|
||
// and the gate text itself shows `## GSTACK REVIEW REPORT` inside a fence too.
|
||
const stripFences = (md: string) => md.replace(/```[\s\S]*?```/g, '');
|
||
|
||
test('gate follows review work, with only the declared cache hook after CEO and Eng verification', () => {
|
||
for (const skill of planSkills) {
|
||
const md = fs.readFileSync(path.join(ROOT, skill, 'SKILL.md'), 'utf-8');
|
||
const stripped = stripFences(md);
|
||
const headings = [...stripped.matchAll(/^## .+$/gm)].map(m => m[0]);
|
||
const lastH2 = headings.at(-1);
|
||
if (['plan-ceo-review', 'plan-eng-review'].includes(skill)) {
|
||
const gate = headings.indexOf('## EXIT PLAN MODE GATE (BLOCKING)');
|
||
expect(gate).toBeGreaterThan(headings.indexOf('## Section self-check (before you finish)'));
|
||
expect(headings.slice(gate)).toEqual(['## EXIT PLAN MODE GATE (BLOCKING)', '## Brain Cache Background Refresh']);
|
||
const tail = md.slice(md.indexOf('## EXIT PLAN MODE GATE (BLOCKING)'));
|
||
const telemetry = tail.indexOf('**Telemetry (run last)**');
|
||
expect(telemetry).toBeGreaterThan(0);
|
||
const cache = tail.indexOf('## Brain Cache Background Refresh');
|
||
if (skill === 'plan-ceo-review') expect(telemetry).toBeGreaterThan(cache);
|
||
else expect(telemetry).toBeLessThan(cache);
|
||
expect(tail).toContain(skill === 'plan-ceo-review' ? 'Passed with a verified persisted report' : 'After the gate passes');
|
||
const finalHandoff = tail.slice(tail.indexOf('## Brain Cache Background Refresh'));
|
||
expect(finalHandoff).toContain(skill === 'plan-ceo-review' ? 'Call ExitPlanMode where required or return to the caller' : 'After success telemetry and cache dispatch, call ExitPlanMode');
|
||
if (skill === 'plan-ceo-review') {
|
||
expect(finalHandoff.indexOf('Call ExitPlanMode')).toBeGreaterThan(finalHandoff.indexOf('**Telemetry (run last)**'));
|
||
expect(finalHandoff).toContain('handoff starts a separate workflow');
|
||
}
|
||
expect(tail).not.toContain('short-circuit when no plan file exists');
|
||
if (skill === 'plan-eng-review') {
|
||
expect(tail.replace(/\s+/g, ' ')).toContain('forbidden report/log persistence or an unrecovered save cannot pass');
|
||
expect(finalHandoff).toContain('only when the host is in plan mode');
|
||
expect(finalHandoff).toContain('Outside plan mode, finish the review in the current conversation; do not call ExitPlanMode');
|
||
} else {
|
||
expect(tail.replace(/\s+/g, ' ')).toContain('**Pass with log-only gaps:**');
|
||
expect(tail.replace(/\s+/g, ' ')).toContain('Label only unwritten artifacts **not persisted**; missing logs do not unsave a verified report');
|
||
expect(tail.replace(/\s+/g, ' ')).toContain('end without success telemetry, ExitPlanMode or the queued handoff');
|
||
}
|
||
} else {
|
||
expect(lastH2, `${skill}/SKILL.md last ## heading (fences stripped)`).toBe('## EXIT PLAN MODE GATE (BLOCKING)');
|
||
}
|
||
if (skill === 'plan-eng-review') {
|
||
const gate = extractMarkdownSection(md, '## EXIT PLAN MODE GATE (BLOCKING)').replace(/\s+/g, ' ');
|
||
expect(gate).toContain('Run this final verification for every review target, in every host mode');
|
||
expect(gate).toContain('If any check fails, follow **Blocked outcome** without success telemetry or ExitPlanMode');
|
||
} else expect(md, `${skill}/SKILL.md gate body`).toContain(skill === 'plan-ceo-review'
|
||
? 'Failed checks use **Gate outcome: Blocked**'
|
||
: 'Failing this gate and calling ExitPlanMode anyway is a contract violation');
|
||
}
|
||
});
|
||
|
||
test('codex/SKILL.md contains gate (mid-file per D5; Step 2B/2C follow)', () => {
|
||
const codex = fs.readFileSync(path.join(ROOT, 'codex', 'SKILL.md'), 'utf-8');
|
||
expect(codex).toContain('## EXIT PLAN MODE GATE (BLOCKING)');
|
||
expect(codex).toContain('Failing this gate and calling ExitPlanMode anyway is a contract violation');
|
||
});
|
||
});
|
||
|
||
describe('scope-gate exceptions drift-guard', () => {
|
||
// The plan-mode auto-select-B exceptions block is hand-duplicated in the
|
||
// plan-eng-review and plan-design-review templates (matching the gate
|
||
// around it, which predates this block). The two copies must stay
|
||
// identical in their target-selection policy. Eng resolves that policy
|
||
// before announcing its target and has a separate bootstrap transport;
|
||
// Design retains its own startup sequence and page vocabulary.
|
||
// A future edit to one copy that silently misses the other fails here
|
||
// instead of drifting. The real fix (shared {{SCOPE_GATE}} resolver) is a
|
||
// filed TODO — this guard is the stopgap that makes the duplication safe.
|
||
const START_MARKER = '**Exceptions — check in this order, BEFORE asking:**';
|
||
const END_MARKER = 'When no exception above applied:';
|
||
|
||
function extractExceptionsBlock(skill: string): string {
|
||
const md = fs.readFileSync(path.join(ROOT, skill, 'SKILL.md'), 'utf-8');
|
||
const start = md.indexOf(START_MARKER);
|
||
expect(start, `${skill}/SKILL.md: exceptions block start marker present`).toBeGreaterThan(-1);
|
||
const end = md.indexOf(END_MARKER, start);
|
||
expect(end, `${skill}/SKILL.md: exceptions block end marker present`).toBeGreaterThan(start);
|
||
return md.slice(start, end + END_MARKER.length);
|
||
}
|
||
|
||
const normalizeTargetPolicy = (block: string) => block
|
||
.split('\n').filter(line => /^[12]\. /.test(line)).join('\n')
|
||
.replace(/ Announce (?:it|an auto-selected plan) in one line so the user can interrupt: "Scope gate: plan mode — auto-selected B \(reviewing <target>\)\."/g, '')
|
||
.replace(' Then run the pre-review audit, mockups, and Step 0 against that plan.', '')
|
||
.replace('a path, a page, a doc they pasted,', 'a path, a doc they pasted,');
|
||
|
||
test('eng and design retain the same target-selection policy across distinct startup sequences', () => {
|
||
const engRaw = extractExceptionsBlock('plan-eng-review');
|
||
const eng = normalizeTargetPolicy(engRaw);
|
||
const design = normalizeTargetPolicy(extractExceptionsBlock('plan-design-review'));
|
||
expect(eng).toBe(design);
|
||
expect(eng).toContain('their choice wins — use it instead');
|
||
expect(eng).toContain('still ambiguous — ask');
|
||
expect(eng).toContain('When in doubt, ask — the gate is the default');
|
||
expect(engRaw.indexOf('their choice wins')).toBeLessThan(engRaw.indexOf('Announce an auto-selected plan'));
|
||
const engMd = fs.readFileSync(path.join(ROOT, 'plan-eng-review/SKILL.md'), 'utf8');
|
||
const transport = engMd.indexOf('Choose listed, enabled MCP AskUserQuestion, otherwise listed native');
|
||
expect(transport).toBeGreaterThan(engMd.indexOf(END_MARKER));
|
||
expect(transport).toBeLessThan(engMd.indexOf('## Preamble (after scope gate)'));
|
||
});
|
||
|
||
test('exceptions block carries the announcement string the PTY detectors pin', () => {
|
||
for (const skill of ['plan-eng-review', 'plan-design-review']) {
|
||
const block = extractExceptionsBlock(skill);
|
||
expect(block, `${skill}: verbatim announcement`).toContain(
|
||
'Scope gate: plan mode — auto-selected B (reviewing <target>).',
|
||
);
|
||
}
|
||
});
|
||
|
||
test('gate menu carries the question strings the PTY question detector pins', () => {
|
||
// isScopeGateQuestionVisible (claude-pty-runner.ts) anchors on the
|
||
// question text + option A's body. If the menu is reworded without
|
||
// updating the detector, the paid smokes' must-stay-false assertions go
|
||
// vacuous — this free pin fails first.
|
||
for (const skill of ['plan-eng-review', 'plan-design-review']) {
|
||
const md = fs.readFileSync(path.join(ROOT, skill, 'SKILL.md'), 'utf-8');
|
||
expect(md, `${skill}: gate question text`).toContain('What should I review?');
|
||
expect(md, `${skill}: option A body text`).toContain('The current branch diff');
|
||
}
|
||
});
|
||
});
|
||
|
||
describe('GSTACK REVIEW REPORT mandatory unresolved-decisions status', () => {
|
||
// Report text rides in PLAN_FILE_REVIEW_REPORT → every report consumer gets it.
|
||
// devex-review is a report consumer but NOT a gate consumer, so the two target
|
||
// sets differ (CP5/CX5). Regression guard: a future token-cut that drops the
|
||
// unresolved-status line again fails here. See plan-flag-unresolved-issues.
|
||
const REPORT_CONSUMERS = [
|
||
'plan-ceo-review',
|
||
'plan-eng-review',
|
||
'plan-design-review',
|
||
'plan-devex-review',
|
||
'codex',
|
||
'devex-review',
|
||
];
|
||
// Gate text rides in EXIT_PLAN_MODE_GATE (lives in SKILL.md, not sections).
|
||
const GATE_SKILLS = [
|
||
'plan-ceo-review',
|
||
'plan-eng-review',
|
||
'plan-design-review',
|
||
'plan-devex-review',
|
||
'codex',
|
||
];
|
||
|
||
for (const skill of REPORT_CONSUMERS) {
|
||
test(`${skill}: report mandates the unresolved-decisions status as final content`, () => {
|
||
const content = readSkillUnion(skill);
|
||
expect(content).toContain('NO UNRESOLVED DECISIONS');
|
||
// The "never omit / always final" contract must be present, not just the phrase.
|
||
expect(content).toContain('Unresolved-decisions status (MANDATORY');
|
||
expect(content).toMatch(skill === 'plan-ceo-review' ? /Never omit this status/ : /never omitted/);
|
||
// \s+ tolerates prose line-wraps within "final non-whitespace line".
|
||
expect(content).toMatch(/final\s+non-whitespace\s+line/);
|
||
});
|
||
}
|
||
|
||
for (const skill of GATE_SKILLS) {
|
||
test(`${skill}: exit gate blocks unless the unresolved status is the final line`, () => {
|
||
const md = fs.readFileSync(path.join(ROOT, skill, 'SKILL.md'), 'utf-8');
|
||
// Gate check #4 — present, sentinel named, and explicitly blocking (no escape).
|
||
expect(md).toContain('NO UNRESOLVED DECISIONS');
|
||
if (['plan-ceo-review', 'plan-eng-review'].includes(skill)) {
|
||
const gate = md.split('## EXIT PLAN MODE GATE (BLOCKING)')[1]!.replace(/\s+/g, ' ');
|
||
expect(gate).toContain('final non-whitespace line is the exact unbolded `NO UNRESOLVED DECISIONS`');
|
||
expect(gate).toContain('or the last bullet under `**UNRESOLVED DECISIONS:**`');
|
||
expect(gate).toMatch(/A bolded sentinel, missing status or (?:any )?trailing prose fails this check/);
|
||
expect(gate).toContain(skill === 'plan-eng-review'
|
||
? 'If any check fails, follow **Blocked outcome** without success telemetry or ExitPlanMode'
|
||
: 'Failed checks use **Gate outcome: Blocked**');
|
||
} else {
|
||
expect(md).toContain('FINAL non-whitespace line is the unresolved-decisions');
|
||
expect(md).toContain('FAILS the gate');
|
||
}
|
||
});
|
||
}
|
||
|
||
test('scripts/resolvers/review.ts source carries the mandatory block + blocking gate', () => {
|
||
const src = fs.readFileSync(path.join(ROOT, 'scripts', 'resolvers', 'review.ts'), 'utf-8');
|
||
// Report resolver: mandatory, never-omitted, exact sentinel, anti-double-count algorithm.
|
||
expect(src).toContain('Unresolved-decisions status (MANDATORY');
|
||
expect(src).toContain('NO UNRESOLVED DECISIONS');
|
||
expect(src).toContain('avoids double-counting');
|
||
expect(src).toContain('DROP the current skill');
|
||
// Gate resolver: the blocking final-line check with no "if applicable" escape.
|
||
expect(src).toContain('FINAL non-whitespace line is the unresolved-decisions');
|
||
expect(src).toContain('FAILS the gate');
|
||
// The old soft wording must be gone from the gate.
|
||
expect(src).not.toContain('absorbs CODEX / CROSS-MODEL / UNRESOLVED lines if applicable');
|
||
});
|
||
});
|
||
|
||
// ─── {{PREAMBLE}} requires an explicit preamble-tier ────────
|
||
|
||
describe('PREAMBLE resolution requires declared preamble-tier', () => {
|
||
test('resolving {{PREAMBLE}} without preamble-tier throws with the template path', async () => {
|
||
const { generatePreamble } = await import('../scripts/resolvers/preamble');
|
||
const { HOST_PATHS } = await import('../scripts/resolvers/types');
|
||
const ctx = {
|
||
skillName: 'tierless-skill',
|
||
tmplPath: 'tierless-skill/SKILL.md.tmpl',
|
||
host: 'claude' as const,
|
||
paths: HOST_PATHS.claude,
|
||
// preambleTier deliberately absent — the generator must refuse to default it.
|
||
};
|
||
expect(() => generatePreamble(ctx)).toThrow(/tierless-skill\/SKILL\.md\.tmpl/);
|
||
expect(() => generatePreamble(ctx)).toThrow(/preamble-tier/);
|
||
});
|
||
|
||
test('every template that resolves {{PREAMBLE}} declares preamble-tier in frontmatter', () => {
|
||
const entries = fs.readdirSync(ROOT, { withFileTypes: true });
|
||
const offenders: string[] = [];
|
||
const checkTmpl = (tmplPath: string) => {
|
||
const tmpl = fs.readFileSync(tmplPath, 'utf-8');
|
||
if (tmpl.includes('{{PREAMBLE}}') && !/^preamble-tier:\s*\d+$/m.test(tmpl)) {
|
||
offenders.push(path.relative(ROOT, tmplPath));
|
||
}
|
||
};
|
||
checkTmpl(path.join(ROOT, 'SKILL.md.tmpl'));
|
||
for (const e of entries) {
|
||
if (!e.isDirectory() || e.name.startsWith('.') || e.name === 'node_modules') continue;
|
||
const tmplPath = path.join(ROOT, e.name, 'SKILL.md.tmpl');
|
||
if (fs.existsSync(tmplPath)) checkTmpl(tmplPath);
|
||
}
|
||
expect(offenders).toEqual([]);
|
||
});
|
||
});
|
||
|
||
// ---------------------------------------------------------------------------
|
||
// #2499: gbrain MCP detection must read BOTH ~/.claude.json scopes.
|
||
// Claude Code registers MCP servers at user scope (.mcpServers) and project
|
||
// scope (.projects["/abs/path"].mcpServers — what `claude mcp add` without
|
||
// --scope user writes). The rendered brain-sync block previously read only
|
||
// user scope, so a correctly configured project-scoped brain was invisible.
|
||
// ---------------------------------------------------------------------------
|
||
describe('brain-sync block reads project-scoped MCP registrations (#2499)', () => {
|
||
// Phase 1: the artifacts-sync bash (including the MCP-scope jq probe) moved
|
||
// from the rendered SKILL.md into bin/gstack-skill-start. Pin the LIVE
|
||
// script bytes — same assertions, new home. The render carries only the
|
||
// ARTIFACTS_SYNC interpretation prose.
|
||
const rendered = fs.readFileSync(path.join(ROOT, 'bin', 'gstack-skill-start'), 'utf-8');
|
||
|
||
test('rendered _GBRAIN_MCP_ENTRY jq resolves project scope with nearest-ancestor cwd match', () => {
|
||
const line = rendered.split('\n').find((l) => l.includes('_GBRAIN_MCP_ENTRY=$('));
|
||
expect(line).toBeDefined();
|
||
// Project-scope read present, driven by $PWD.
|
||
expect(line!).toContain('--arg cwd "$PWD"');
|
||
expect(line!).toContain('.projects');
|
||
// User scope still resolved first.
|
||
expect(line!).toContain('.mcpServers.gbrain');
|
||
// The old user-scope-only filter is gone from the rendered output.
|
||
expect(rendered).not.toContain('.mcpServers.gbrain.type // .mcpServers.gbrain.transport');
|
||
expect(rendered).not.toContain(".mcpServers.gbrain.url // empty");
|
||
});
|
||
|
||
test('rendered _GBRAIN_MCP_TYPE and _GBRAIN_HOST extract from the resolved entry', () => {
|
||
const typeLine = rendered.split('\n').find((l) => l.includes('_GBRAIN_MCP_TYPE=$('));
|
||
const hostLine = rendered.split('\n').find((l) => l.includes('_GBRAIN_HOST=$('));
|
||
expect(typeLine).toBeDefined();
|
||
expect(hostLine).toBeDefined();
|
||
expect(typeLine!).toContain('_GBRAIN_MCP_ENTRY');
|
||
expect(hostLine!).toContain('_GBRAIN_MCP_ENTRY');
|
||
});
|
||
|
||
test('rendered jq lines FUNCTION: project-scoped registration resolves for a cwd inside the project', () => {
|
||
// Execute the exact rendered bytes, not a re-derivation: extract the
|
||
// _GBRAIN_MCP_ENTRY + _GBRAIN_MCP_TYPE lines from the generated SKILL.md
|
||
// and run them in bash against a fixture ~/.claude.json that carries ONLY
|
||
// a project-scoped gbrain registration.
|
||
const lines = rendered.split('\n');
|
||
const entryLine = lines.find((l) => l.includes('_GBRAIN_MCP_ENTRY=$('));
|
||
const typeLine = lines.find((l) => l.includes('_GBRAIN_MCP_TYPE=$('));
|
||
expect(entryLine).toBeDefined();
|
||
expect(typeLine).toBeDefined();
|
||
|
||
const tmpHome = fs.mkdtempSync(path.join(os.tmpdir(), 'gstack-2499-home-'));
|
||
const projectDir = fs.mkdtempSync(path.join(os.tmpdir(), 'gstack-2499-proj-'));
|
||
const nestedCwd = path.join(projectDir, 'src', 'deep');
|
||
fs.mkdirSync(nestedCwd, { recursive: true });
|
||
try {
|
||
fs.writeFileSync(
|
||
path.join(tmpHome, '.claude.json'),
|
||
JSON.stringify({
|
||
projects: {
|
||
[projectDir]: {
|
||
mcpServers: { gbrain: { type: 'http', url: 'https://brain.example.com/mcp' } },
|
||
},
|
||
},
|
||
}),
|
||
);
|
||
const script = `cd "$1" || exit 1\n${entryLine!.trim()}\n${typeLine!.trim()}\necho "RESOLVED:$_GBRAIN_MCP_TYPE"`;
|
||
const r = spawnSync('bash', ['-c', script, 'bash', nestedCwd], {
|
||
encoding: 'utf-8',
|
||
env: { ...process.env, HOME: tmpHome },
|
||
timeout: 10_000,
|
||
});
|
||
expect(r.stdout).toContain('RESOLVED:http');
|
||
|
||
// Discriminator: a cwd OUTSIDE the project must NOT resolve it.
|
||
const outside = spawnSync('bash', ['-c', script, 'bash', os.tmpdir()], {
|
||
encoding: 'utf-8',
|
||
env: { ...process.env, HOME: tmpHome },
|
||
timeout: 10_000,
|
||
});
|
||
expect(outside.stdout).toContain('RESOLVED:\n');
|
||
} finally {
|
||
fs.rmSync(tmpHome, { recursive: true, force: true });
|
||
fs.rmSync(projectDir, { recursive: true, force: true });
|
||
}
|
||
});
|
||
});
|