Files
gstack/test/skill-cross-model-recommendation-emit.test.ts
T
Garry TanandClaude Opus 4.7 4ab0269729 feat(codex+review): require synthesis Recommendation in cross-model skills
Extends the v1.25.1.0 AskUserQuestion recommendation-quality coverage to the
cross-model synthesis surfaces that were previously emitting prose without a
structured recommendation:

- /codex review (Step 2A) — after presenting Codex output + GATE verdict,
  must emit `Recommendation: <action> because <reason>` line. Reason must
  compare against alternatives (other findings, fix-vs-ship, fix-order).
- /codex challenge (Step 2B) — same requirement after adversarial output.
- /codex consult (Step 2C) — same requirement after consult presentation,
  with examples for plan-review consults that engage with specific Codex
  insights.
- Claude adversarial subagent (scripts/resolvers/review.ts:446, used by
  /ship Step 11 + standalone /review) — subagent prompt now ends with
  "After listing findings, end your output with ONE line in the canonical
  format Recommendation: <action> because <reason>". Codex adversarial
  command (line 461) gets the same final-line requirement.

The same `judgeRecommendation` helper grades both AskUserQuestion and
cross-model synthesis — one rubric, two surfaces. Substance-5 cross-model
recommendations explicitly compare against alternatives (a different
finding, fix-vs-ship, fix-order). Generic synthesis ("because adversarial
review found things") fails at threshold ≥ 4.

Tests:
- test/llm-judge-recommendation.test.ts gains 5 cross-model fixtures (3
  substance ≥ 4, 2 substance < 4). Existing rubric correctly grades them.
- test/skill-cross-model-recommendation-emit.test.ts (new, free-tier) —
  static guard greps codex/SKILL.md.tmpl + scripts/resolvers/review.ts for
  the canonical emit instruction. Trips before any paid eval if the
  templates drift.

Touchfile: extended `llm-judge-recommendation` entry with codex/SKILL.md.tmpl
and scripts/resolvers/review.ts so synthesis-template edits invalidate the
fixture re-run.

Verified: free `bun test` exits 0 (5/5 static emit-guard tests pass), paid
fixture passes 45/45 expect calls in 24s with the cross-model substance-5
fixtures correctly judged at >= 4.

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-05-01 19:38:12 -07:00

70 lines
3.6 KiB
TypeScript

/**
* Static guard for cross-model synthesis recommendation emit instructions.
*
* v1.25.1.0+ extended the AskUserQuestion recommendation-quality coverage
* to cross-model skills (/codex review/challenge/consult, the Claude
* adversarial subagent, and the Codex adversarial pass). Each surface MUST
* tell the model to end its synthesis with a canonical
* `Recommendation: <action> because <reason>`
* line so judgeRecommendation can grade it (see test/llm-judge-recommendation
* for the rubric exercise).
*
* Free, deterministic, single-purpose: if any contributor edits these
* templates and removes the emit instruction, this test trips before the
* change reaches a paid eval. The runtime grading still happens via
* judgeRecommendation when the skills run for real; this test just pins the
* source of truth.
*/
import { describe, test, expect } from 'bun:test';
import * as fs from 'fs';
import * as path from 'path';
const ROOT = path.resolve(import.meta.dir, '..');
describe('cross-model synthesis emit instructions', () => {
test('codex/SKILL.md.tmpl Step 2A (review) requires a synthesis Recommendation', () => {
const tmpl = fs.readFileSync(path.join(ROOT, 'codex', 'SKILL.md.tmpl'), 'utf-8');
const step2a = sliceBetween(tmpl, '## Step 2A:', '## Step 2B:');
expect(step2a, 'Step 2A section not found in codex template').not.toBe('');
expect(step2a).toMatch(/Synthesis recommendation \(REQUIRED\)/);
expect(step2a).toMatch(/Recommendation:\s*<action>\s*because/);
});
test('codex/SKILL.md.tmpl Step 2B (challenge) requires a synthesis Recommendation', () => {
const tmpl = fs.readFileSync(path.join(ROOT, 'codex', 'SKILL.md.tmpl'), 'utf-8');
const step2b = sliceBetween(tmpl, '## Step 2B:', '## Step 2C:');
expect(step2b, 'Step 2B section not found in codex template').not.toBe('');
expect(step2b).toMatch(/Synthesis recommendation \(REQUIRED\)/);
expect(step2b).toMatch(/Recommendation:\s*<action>\s*because/);
});
test('codex/SKILL.md.tmpl Step 2C (consult) requires a synthesis Recommendation', () => {
const tmpl = fs.readFileSync(path.join(ROOT, 'codex', 'SKILL.md.tmpl'), 'utf-8');
const step2c = sliceBetween(tmpl, '## Step 2C:', '## Model & Reasoning');
expect(step2c, 'Step 2C section not found in codex template').not.toBe('');
expect(step2c).toMatch(/Synthesis recommendation \(REQUIRED\)/);
expect(step2c).toMatch(/Recommendation:\s*<action>\s*because/);
});
test('scripts/resolvers/review.ts Claude adversarial subagent prompt requires Recommendation', () => {
const resolver = fs.readFileSync(path.join(ROOT, 'scripts', 'resolvers', 'review.ts'), 'utf-8');
// The Claude subagent prompt must instruct the model to emit a final
// canonical Recommendation line.
expect(resolver).toMatch(/Claude adversarial subagent[\s\S]+?Recommendation:\s*<action>\s*because/);
});
test('scripts/resolvers/review.ts Codex adversarial command requires Recommendation', () => {
const resolver = fs.readFileSync(path.join(ROOT, 'scripts', 'resolvers', 'review.ts'), 'utf-8');
// The codex exec command's prompt string must include the emit
// instruction. Match within the codex adversarial section.
expect(resolver).toMatch(/Codex adversarial challenge[\s\S]+?Recommendation:\s*<action>\s*because/);
});
});
function sliceBetween(text: string, startMarker: string, endMarker: string): string {
const start = text.indexOf(startMarker);
if (start < 0) return '';
const end = text.indexOf(endMarker, start + startMarker.length);
return end > start ? text.slice(start, end) : text.slice(start);
}