mirror of
https://github.com/garrytan/gstack.git
synced 2026-10-03 01:46:55 +02:00
test: clean up the paid eval lane (B1-B4, B6, B7)
- B1: delete paid files that assert nothing or cannot pass meaningfully: skill-llm-eval-spec and skill-e2e-spec-execute (test.todo), gemini-e2e (+ gemini-session-runner; no gemini CLI in CI), ship-idempotency (red since v1.63), the two opus-4-7 *-sonnet overlay wrappers, conductor-prose (+ its source-evaluation replay), codex-e2e-plan-format; drop their keys, scripts and census rows. - B2: skill-llm-eval grades browse/sections/command-list.md with one union judge that also carries the baseline score pin; regression-vs-baseline deleted (paid run: pass, c4/c4/a4). - B3: memory-pipeline, ios-qa, ios-qa-swift-build and plan-tune-cathedral make no model calls; renamed out of the paid glob so they run on every PR. Swift builds need GSTACK_TEST_SWIFT=1; device stub deleted. - B4: codex-e2e*, outside-voice, aside and ios-device cannot run in the CI image; excluded from the weekly lane with a tracked re-entry condition. - B6: fold opus-47's negative routing controls into skill-routing-e2e journey-negatives (paid run: 3/3 unrouted) and delete the file. - B7: delete the never-green brain-privacy-gate eval; a free gstack-skill-start test now proves consent precedes artifacts egress.
This commit is contained in:
1 parent
5d032ef299
commit
53e7f3212f
58 files changed
+273
-2594
No files matched your search
+21
-199
@@ -12,7 +12,6 @@
|
||||
|
||||
import { afterAll, expect } from 'bun:test';
|
||||
import { JUDGE_MS } from './helpers/eval-budgets';
|
||||
import Anthropic from '@anthropic-ai/sdk';
|
||||
import * as fs from 'fs';
|
||||
import * as path from 'path';
|
||||
import { callJudge, judge, JudgeRefusalError, DEFAULT_JUDGE_MAX_TOKENS } from './helpers/llm-judge';
|
||||
@@ -62,12 +61,11 @@ function readBrowseCommandSection(): string {
|
||||
}
|
||||
|
||||
/** Slice a section out of the command-list section file, guarded non-empty. */
|
||||
function sliceBrowseSection(startHeader: string, endHeader?: string): string {
|
||||
function sliceBrowseSection(startHeader: string): string {
|
||||
const content = readBrowseCommandSection();
|
||||
const start = content.indexOf(startHeader);
|
||||
if (start < 0) throw new Error(`browse/sections/command-list.md: "${startHeader}" not found`);
|
||||
const end = endHeader ? content.indexOf(endHeader) : -1;
|
||||
const section = end > start ? content.slice(start, end) : content.slice(start);
|
||||
const section = content.slice(start);
|
||||
if (section.trim().length < 200) {
|
||||
throw new Error(`browse/sections/command-list.md slice at "${startHeader}" is empty/stub — regenerate with: bun run gen:skill-docs`);
|
||||
}
|
||||
@@ -91,85 +89,45 @@ function testIfSelected(testName: string, fn: () => Promise<void>, timeout: numb
|
||||
}
|
||||
|
||||
describeIfSelected('LLM-as-judge quality evals', [
|
||||
'command reference table', 'snapshot flags reference',
|
||||
'browse/SKILL.md reference', 'setup block', 'regression vs baseline',
|
||||
'browse/SKILL.md reference', 'setup block',
|
||||
], () => {
|
||||
testIfSelected('command reference table', async () => {
|
||||
const t0 = Date.now();
|
||||
// Browse carve: the command reference lives in the generated on-demand
|
||||
// section browse/sections/command-list.md now (read via non-empty guard).
|
||||
const section = sliceBrowseSection('## Full Command List');
|
||||
|
||||
const scores = await judge('command reference table', section);
|
||||
console.log('Command reference scores:', JSON.stringify(scores, null, 2));
|
||||
|
||||
// Completeness threshold is 3 (not 4) — the command reference table is
|
||||
// intentionally terse (quick-reference format). The judge consistently scores
|
||||
// completeness=3 because detailed argument docs live in per-command sections.
|
||||
evalCollector?.addTest({
|
||||
name: 'command reference table',
|
||||
suite: 'LLM-as-judge quality evals',
|
||||
tier: 'llm-judge',
|
||||
passed: scores.clarity >= 3 && scores.completeness >= 3 && scores.actionability >= 4,
|
||||
duration_ms: Date.now() - t0,
|
||||
cost_usd: 0.02,
|
||||
judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability },
|
||||
judge_reasoning: scores.reasoning,
|
||||
});
|
||||
|
||||
expect(scores.clarity).toBeGreaterThanOrEqual(3);
|
||||
expect(scores.completeness).toBeGreaterThanOrEqual(3);
|
||||
expect(scores.actionability).toBeGreaterThanOrEqual(4);
|
||||
}, JUDGE_MS);
|
||||
|
||||
testIfSelected('snapshot flags reference', async () => {
|
||||
const t0 = Date.now();
|
||||
// Browse carve: snapshot flags live in browse/sections/command-list.md now,
|
||||
// ordered before '## Full Command List' (the '## CSS Inspector' end boundary
|
||||
// stayed in the skeleton).
|
||||
const section = sliceBrowseSection('## Snapshot Flags', '## Full Command List');
|
||||
|
||||
const scores = await judge('snapshot flags reference', section);
|
||||
console.log('Snapshot flags scores:', JSON.stringify(scores, null, 2));
|
||||
|
||||
evalCollector?.addTest({
|
||||
name: 'snapshot flags reference',
|
||||
suite: 'LLM-as-judge quality evals',
|
||||
tier: 'llm-judge',
|
||||
passed: scores.clarity >= 3 && scores.completeness >= 4 && scores.actionability >= 4,
|
||||
duration_ms: Date.now() - t0,
|
||||
cost_usd: 0.02,
|
||||
judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability },
|
||||
judge_reasoning: scores.reasoning,
|
||||
});
|
||||
|
||||
expect(scores.clarity).toBeGreaterThanOrEqual(3);
|
||||
expect(scores.completeness).toBeGreaterThanOrEqual(4);
|
||||
expect(scores.actionability).toBeGreaterThanOrEqual(4);
|
||||
}, JUDGE_MS);
|
||||
|
||||
testIfSelected('browse/SKILL.md reference', async () => {
|
||||
const t0 = Date.now();
|
||||
// Browse carve: flags + commands are the whole generated section file.
|
||||
// Browse carve: snapshot flags + the full command list are the whole
|
||||
// generated section file; one judge grades the union. Scores are also
|
||||
// pinned against test/fixtures/eval-baselines.json (UPDATE_BASELINES=1
|
||||
// rewrites the pin).
|
||||
const section = sliceBrowseSection('## Snapshot Flags');
|
||||
|
||||
const scores = await judge('browse skill reference (flags + commands)', section);
|
||||
console.log('Browse SKILL.md scores:', JSON.stringify(scores, null, 2));
|
||||
|
||||
const baselinesPath = path.join(ROOT, 'test', 'fixtures', 'eval-baselines.json');
|
||||
const baselines = JSON.parse(fs.readFileSync(baselinesPath, 'utf-8'));
|
||||
const regressions = (['clarity', 'completeness', 'actionability'] as const)
|
||||
.filter(dim => scores[dim] < baselines.browse_skill[dim])
|
||||
.map(dim => `browse_skill.${dim}: ${scores[dim]} < baseline ${baselines.browse_skill[dim]}`);
|
||||
if (process.env.UPDATE_BASELINES) {
|
||||
baselines.browse_skill = { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability };
|
||||
fs.writeFileSync(baselinesPath, JSON.stringify(baselines, null, 2) + '\n');
|
||||
console.log('Updated eval baselines');
|
||||
}
|
||||
|
||||
evalCollector?.addTest({
|
||||
name: 'browse/SKILL.md reference',
|
||||
suite: 'LLM-as-judge quality evals',
|
||||
tier: 'llm-judge',
|
||||
passed: scores.clarity >= 3 && scores.completeness >= 4 && scores.actionability >= 4,
|
||||
passed: scores.clarity >= 3 && scores.completeness >= 4 && scores.actionability >= 4 && regressions.length === 0,
|
||||
duration_ms: Date.now() - t0,
|
||||
cost_usd: 0.02,
|
||||
judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability },
|
||||
judge_reasoning: scores.reasoning,
|
||||
judge_reasoning: regressions.length ? `${scores.reasoning} | ${regressions.join('; ')}` : scores.reasoning,
|
||||
});
|
||||
|
||||
expect(scores.clarity).toBeGreaterThanOrEqual(3);
|
||||
expect(scores.completeness).toBeGreaterThanOrEqual(4);
|
||||
expect(scores.actionability).toBeGreaterThanOrEqual(4);
|
||||
expect(regressions).toEqual([]);
|
||||
}, JUDGE_MS);
|
||||
|
||||
testIfSelected('setup block', async () => {
|
||||
@@ -205,89 +163,6 @@ describeIfSelected('LLM-as-judge quality evals', [
|
||||
expect(scores.clarity).toBeGreaterThanOrEqual(3);
|
||||
}, JUDGE_MS);
|
||||
|
||||
testIfSelected('regression vs baseline', async () => {
|
||||
const t0 = Date.now();
|
||||
// Browse carve: the command reference lives in browse/sections/command-list.md.
|
||||
const genSection = sliceBrowseSection('## Full Command List');
|
||||
|
||||
const baseline = `## Command Reference
|
||||
|
||||
### Navigation
|
||||
| Command | Description |
|
||||
|---------|-------------|
|
||||
| \`goto <url>\` | Navigate to URL |
|
||||
| \`back\` / \`forward\` | History navigation |
|
||||
| \`reload\` | Reload page |
|
||||
| \`url\` | Print current URL |
|
||||
|
||||
### Interaction
|
||||
| Command | Description |
|
||||
|---------|-------------|
|
||||
| \`click <sel>\` | Click element |
|
||||
| \`fill <sel> <val>\` | Fill input |
|
||||
| \`select <sel> <val>\` | Select dropdown |
|
||||
| \`hover <sel>\` | Hover element |
|
||||
| \`type <text>\` | Type into focused element |
|
||||
| \`press <key>\` | Press key (Enter, Tab, Escape) |
|
||||
| \`scroll [sel]\` | Scroll element into view |
|
||||
| \`wait <sel>\` | Wait for element (max 10s) |
|
||||
| \`wait --networkidle\` | Wait for network to be idle |
|
||||
| \`wait --load\` | Wait for page load event |
|
||||
|
||||
### Inspection
|
||||
| Command | Description |
|
||||
|---------|-------------|
|
||||
| \`js <expr>\` | Run JavaScript |
|
||||
| \`css <sel> <prop>\` | Computed CSS |
|
||||
| \`attrs <sel>\` | Element attributes |
|
||||
| \`is <prop> <sel>\` | State check (visible/hidden/enabled/disabled/checked/editable/focused) |
|
||||
| \`console [--clear\\|--errors]\` | Console messages (--errors filters to error/warning) |`;
|
||||
|
||||
const client = new Anthropic();
|
||||
const response = await client.messages.create({
|
||||
model: 'claude-sonnet-4-6',
|
||||
max_tokens: 1024,
|
||||
messages: [{
|
||||
role: 'user',
|
||||
content: `You are comparing two versions of CLI documentation for an AI coding agent.
|
||||
|
||||
VERSION A (baseline — hand-maintained):
|
||||
${baseline}
|
||||
|
||||
VERSION B (auto-generated from source):
|
||||
${genSection}
|
||||
|
||||
Which version is better for an AI agent trying to use these commands? Consider:
|
||||
- Completeness (more commands documented? all args shown?)
|
||||
- Clarity (descriptions helpful?)
|
||||
- Coverage (missing commands in either version?)
|
||||
|
||||
Respond with ONLY valid JSON:
|
||||
{"winner": "A" or "B" or "tie", "reasoning": "brief explanation", "a_score": N, "b_score": N}
|
||||
|
||||
Scores are 1-5 overall quality.`,
|
||||
}],
|
||||
});
|
||||
|
||||
const text = response.content[0].type === 'text' ? response.content[0].text : '';
|
||||
const jsonMatch = text.match(/\{[\s\S]*\}/);
|
||||
if (!jsonMatch) throw new Error(`Judge returned non-JSON: ${text.slice(0, 200)}`);
|
||||
const result = JSON.parse(jsonMatch[0]);
|
||||
console.log('Regression comparison:', JSON.stringify(result, null, 2));
|
||||
|
||||
evalCollector?.addTest({
|
||||
name: 'regression vs baseline',
|
||||
suite: 'LLM-as-judge quality evals',
|
||||
tier: 'llm-judge',
|
||||
passed: result.b_score >= result.a_score,
|
||||
duration_ms: Date.now() - t0,
|
||||
cost_usd: 0.02,
|
||||
judge_scores: { a_score: result.a_score, b_score: result.b_score },
|
||||
judge_reasoning: result.reasoning,
|
||||
});
|
||||
|
||||
expect(result.b_score).toBeGreaterThanOrEqual(result.a_score);
|
||||
}, JUDGE_MS);
|
||||
});
|
||||
|
||||
// --- Part 7: QA skill quality evals (C6) ---
|
||||
@@ -523,59 +398,6 @@ score (1-5): 5 = perfectly consistent, 1 = contradictory`);
|
||||
}, JUDGE_MS);
|
||||
});
|
||||
|
||||
// --- Part 7: Baseline score pinning (C9) ---
|
||||
|
||||
describeIfSelected('Baseline score pinning', ['baseline score pinning'], () => {
|
||||
const baselinesPath = path.join(ROOT, 'test', 'fixtures', 'eval-baselines.json');
|
||||
|
||||
testIfSelected('baseline score pinning', async () => {
|
||||
const t0 = Date.now();
|
||||
if (!fs.existsSync(baselinesPath)) {
|
||||
console.log('No baseline file found — skipping pinning check');
|
||||
return;
|
||||
}
|
||||
|
||||
const baselines = JSON.parse(fs.readFileSync(baselinesPath, 'utf-8'));
|
||||
const regressions: string[] = [];
|
||||
|
||||
// Browse carve: the command reference lives in browse/sections/command-list.md.
|
||||
const cmdSection = sliceBrowseSection('## Full Command List');
|
||||
const cmdScores = await judge('command reference table', cmdSection);
|
||||
|
||||
for (const dim of ['clarity', 'completeness', 'actionability'] as const) {
|
||||
if (cmdScores[dim] < baselines.command_reference[dim]) {
|
||||
regressions.push(`command_reference.${dim}: ${cmdScores[dim]} < baseline ${baselines.command_reference[dim]}`);
|
||||
}
|
||||
}
|
||||
|
||||
if (process.env.UPDATE_BASELINES) {
|
||||
baselines.command_reference = {
|
||||
clarity: cmdScores.clarity,
|
||||
completeness: cmdScores.completeness,
|
||||
actionability: cmdScores.actionability,
|
||||
};
|
||||
fs.writeFileSync(baselinesPath, JSON.stringify(baselines, null, 2) + '\n');
|
||||
console.log('Updated eval baselines');
|
||||
}
|
||||
|
||||
const passed = regressions.length === 0;
|
||||
evalCollector?.addTest({
|
||||
name: 'baseline score pinning',
|
||||
suite: 'Baseline score pinning',
|
||||
tier: 'llm-judge',
|
||||
passed,
|
||||
duration_ms: Date.now() - t0,
|
||||
cost_usd: 0.02,
|
||||
judge_scores: { clarity: cmdScores.clarity, completeness: cmdScores.completeness, actionability: cmdScores.actionability },
|
||||
judge_reasoning: passed ? 'All scores at or above baseline' : regressions.join('; '),
|
||||
});
|
||||
|
||||
if (!passed) {
|
||||
throw new Error(`Score regressions detected:\n${regressions.join('\n')}`);
|
||||
}
|
||||
}, JUDGE_MS);
|
||||
});
|
||||
|
||||
// --- Workflow SKILL.md quality evals (10 new tests for 100% coverage) ---
|
||||
|
||||
/**
|
||||
|
||||
Reference in new issue
Block a user