Files
gstack/test/coverage-audit-evidence.test.ts
T
garrytan 7cbb5e04e4 test: derive paid touchfiles from each eval's static closure (E)
touchfiles.test.ts now checks, per key, that the paid file's static
test/helpers and test/fixtures closure (plus fixture paths it names in string
literals) is covered, and names the file, path, chain and key to fix when it is
not. Free *.test.ts files are no longer touchfiles, so editing a free replay
test stops selecting paid evals: 950 entries removed, 653 real closure paths
added. The hand-copied inventories go: periodic-fixture-selection,
fake-impeccable-touchfiles and 45 per-file selection examples. Selection for
the sample edits (plan-eng-review template, claude-pty-runner,
plan-count-fixture, gstack-config) loses no case under either profile.
CONTRIBUTING documents the rule and its lower bound.
2026-09-29 07:00:28 +00:00

986 lines
69 KiB
TypeScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
import { describe, expect, test } from 'bun:test';
import * as path from 'node:path';
import fixture from './fixtures/coverage-audit-ae.json';
import ciDiagrams from './fixtures/coverage-audit-ci-diagrams.json';
import { coverageAuditVerdict } from './helpers/coverage-audit-evidence';
import { recordE2E } from './helpers/e2e-helpers';
import fixture_coverage_audit_af from './fixtures/coverage-audit-af.json';
import { posix } from 'node:path';
import { win32 } from 'node:path';
import { coverageAuditReadEvidence } from './helpers/coverage-audit-evidence';
import fixture_coverage_audit_aw from './fixtures/coverage-audit-aw.json';
import captured_coverage_audit_shell_legend_at from './fixtures/coverage-audit-shell-legend-at.json';
import fixture_coverage_checkbox_tail_av from './fixtures/coverage-checkbox-tail-av.json';
import captured_coverage_diagram_legend_as from './fixtures/coverage-diagram-legend-as.json';
import fixture_coverage_shell_display_aq from './fixtures/coverage-shell-display-aq.json';
import billing_coverage_shell_display_aq from './fixtures/coverage-audit-ae.json';
const clone = <T>(v:T):T => structuredClone(v);
const diagram = '```text\nsrc/billing.ts\n├── processPayment: happy path [TESTED]\n└── refundPayment [UNTESTED]\n```';
function synthetic() {
const cwd = '/tmp/coverage-audit-evidence-owned';
const files = {cwd, source:{path:path.join(cwd,'src/billing.ts'),content:fixture.files.source},
tests:{path:path.join(cwd,'test/billing.test.ts'),content:fixture.files.tests}};
const transcript:any[] = [{type:'system',subtype:'init',session_id:'parent',cwd}];
for (const [id,file] of Object.entries({source:files.source,tests:files.tests})) {
transcript.push({type:'assistant',session_id:'parent',parent_tool_use_id:null,message:{role:'assistant',content:[
{type:'tool_use',id,name:'Read',input:{file_path:file.path}},
]}});
transcript.push({type:'user',session_id:'parent',parent_tool_use_id:null,message:{role:'user',content:[
{type:'tool_result',tool_use_id:id,content:file.content},
]}});
}
return {files,result:{exitReason:'success',browseErrors:[],output:diagram,transcript} as any};
}
const verdict = (s:ReturnType<typeof synthetic>) => coverageAuditVerdict(s.result,s.files);
const block = (s:ReturnType<typeof synthetic>,i:number) => s.result.transcript[i].message.content[0];
describe('coverage audit native evidence',()=>{
test('literal cat operands preserve quoted whitespace for single and multiple owned paths', () => {
for (const quote of ["'", '"']) for (const flags of ['', '-n ', '-n -- ']) {
const s = synthetic();
s.files.cwd = '/repo with space';
s.files.source.path = s.files.cwd + '/src/billing source.ts';
s.files.tests.path = s.files.cwd + '/test/billing test.ts';
s.result.transcript[0].cwd = s.files.cwd;
for (const [i, file] of [[1, s.files.source], [3, s.files.tests]] as const) {
Object.assign(block(s, i), { name: 'Bash', input: { command: `cat ${flags}${quote}${file.path}${quote}` } });
}
expect(verdict(s)).toMatchObject({ sourceRead: true, testsRead: true });
block(s, 1).input.command = `cat ${flags}${quote}${s.files.source.path}${quote} ${quote}${s.files.tests.path}${quote}`;
block(s, 2).content = s.files.source.content + '\n' + s.files.tests.content;
s.result.transcript.splice(3);
expect(verdict(s)).toMatchObject({ sourceRead: true, testsRead: true });
for (const operands of [
`${quote}${s.files.source.path}${quote}suffix`,
`${quote}${s.files.source.path}${quote}${quote}${s.files.tests.path}${quote}`,
`${quote}${s.files.source.path}`, '"$SOURCE_FILE"', '`cat path`',
]) {
block(s, 1).input.command = `cat ${flags}${operands}`;
expect(verdict(s), operands).toMatchObject({ sourceRead: false, testsRead: false });
}
}
});
test('single word legend entries consume complete unqualified coverage and quality clauses', () => {
const s = synthetic();
const output = (legend: string) => '```text\nprocessPayment()\n└─ happy path [OK]\nrefundPayment()\n└─ happy path [GAP]\n' + legend + '\n```';
for (const legend of [
'Legend: [OK] covered\nLegend: [GAP] no test',
'Legend: ★★★ edges + errors ★★ happy path only ★ smoke [OK] tested\nLegend: [GAP] no test [→E2E] recommend integration test',
]) {
s.result.output = output(legend);
expect(verdict(s).diagram, legend).toBe(true);
for (const qualified of [
legend.replace('[OK] covered', '[OK] covered only if approved').replace('[OK] tested', '[OK] tested only if approved'),
legend.replace('[GAP] no test', '[GAP] no test except refunds'),
legend.replace('[GAP] no test', '[GAP] no test unless approved'),
legend.replace('[GAP] no test', 'hypothetical [GAP] no test'),
legend.replace('[OK]', 'not [OK]'),
legend + ' unknown qualifier',
]) {
s.result.output = output(qualified);
expect(verdict(s).diagram, qualified).toBe(false);
}
}
});
test('all four exact completed public attempts delivered both files and the seeded diagram',()=>{
expect(fixture.provenance.actualPassedCases).toBe(0);
for(const row of fixture.rows){
const files={cwd:row.cwd,source:{path:path.join(row.cwd,'src/billing.ts'),content:fixture.files.source},
tests:{path:path.join(row.cwd,'test/billing.test.ts'),content:fixture.files.tests}};
expect(coverageAuditVerdict(row.result as any,files)).toEqual({sourceRead:true,testsRead:true,diagram:true,passed:true,failures:[]});
}
});
test('both exact CI diagrams retain covered payment and missing refund paths', () => {
expect(ciDiagrams.provenance.recordedAttemptOutcomes).toEqual(['failed', 'failed']);
expect(ciDiagrams.provenance.paidOutcomesReclassified).toBe(false);
for (const row of ciDiagrams.diagrams) {
const s = synthetic(); s.result.output = row.text;
expect(verdict(s)).toEqual({ sourceRead: true, testsRead: true, diagram: true, passed: true, failures: [] });
}
});
const displayLegend = (legend: string, covered = '#', gap = ' ') => '```text\n' + legend + '\n'
+ 'processPayment(amount, currency)\n└─ valid return success [' + covered + ']\n'
+ 'refundPayment(paymentId, reason)\n└─ valid return refunded [' + gap + '] GAP\n```';
test('paid coverage diagrams accept a branch line without an arrowhead and a declared hash checkbox', () => {
for (const output of [
displayLegend('Legend: [✓] tested [✗] GAP (no test) ── branch', '✓', '✗'),
displayLegend('src/billing.ts — coverage map [#] tested [ ] GAP'),
displayLegend('Legend: [#] tested [ ] no test'),
displayLegend('src/billing.ts — coverage map [x] tested [ ] GAP', 'x'),
]) {
const s = synthetic(); s.result.output = output;
expect(verdict(s)).toEqual({ sourceRead: true, testsRead: true, diagram: true, passed: true, failures: [] });
}
});
// Exact public diagram from the completed, failed September 21 /review run.
// Its footer declares both symbol meanings without punctuation after Legend.
const symbolFooterDiagram = `\`\`\`
src/billing.ts test/billing.test.ts
======================================================================================
processPayment(amount, currency) describe('processPayment')
│
├── [✓] amount > 0 && currency in {USD, EUR} processes valid payment (L6-9)
│ → return { status: 'success', ... } (L5) processPayment(100, 'USD')
│
├── [✗] amount <= 0 ── NO TEST ──
│ → throw 'Invalid amount' (L3) gap: 0, negative values untested
│
└── [✗] currency not USD/EUR ── NO TEST ──
→ throw 'Unsupported currency' (L4) gap: 'GBP', '', lowercase 'usd'
refundPayment(paymentId, reason) (no describe block; not imported)
│
├── [✗] paymentId && reason truthy ── NO TEST ──
│ → return { status: 'refunded', ... } (L11) gap: happy path never exercised
│
├── [✗] !paymentId ── NO TEST ──
│ → throw 'Payment ID required' (L9) gap: '' / undefined untested
│
└── [✗] !reason ── NO TEST ──
→ throw 'Reason required' (L10) gap: '' / undefined untested
======================================================================================
Legend [✓] covered [✗] gap
Branches: 1 / 6 covered (17%)
Functions: 1 / 2 covered (50%)
Guard clauses tested: 0 / 4
\`\`\``;
test('the exact paid symbol footer may omit its colon', () => {
const s = synthetic(); s.result.output = symbolFooterDiagram;
expect(verdict(s)).toEqual({ sourceRead: true, testsRead: true, diagram: true, passed: true, failures: [] });
});
test('a colonless symbol footer still requires a current, affirmative, owned key', () => {
const key = 'Legend [✓] covered [✗] gap';
for (const replacement of ['', '> ' + key, '"' + key + '"', 'Example: ' + key,
'If approved: ' + key, key.replace('covered [✗] gap', 'gap [✗] covered'),
key.replace('[✗] gap', '[✗] covered'), key.replace('[✗]', '[✓]'),
key + ' except refunds', key + '\nLegend [✓] gap [✗] covered',
key + '\nThis legend is withdrawn.', key + '\nThis legend applies only if approved.',
]) {
const s = synthetic(); s.result.output = symbolFooterDiagram.replace(key, replacement);
expect(verdict(s).diagram, replacement).toBe(false);
}
for (const output of [
'\`\`\`text\n' + key + '\n\`\`\`\n' + symbolFooterDiagram.replace(key, ''),
symbolFooterDiagram.replaceAll('refundPayment', 'otherRefund'),
symbolFooterDiagram.replace('├── [✓] amount', '├── [✗] amount'),
'Example:\n' + symbolFooterDiagram, '\`\`\`\`markdown\n' + symbolFooterDiagram + '\n\`\`\`\`',
]) {
const s = synthetic(); s.result.output = output; expect(verdict(s).diagram).toBe(false);
}
});
test('hash checkbox and branch-line legends retain explicit local meanings and ownership', () => {
const caption = 'src/billing.ts — coverage map [#] tested [ ] GAP';
for (const legend of ['', '> ' + caption, '"' + caption + '"', 'Example: ' + caption,
'If approved: ' + caption, caption.replace('[#] tested [ ] GAP', '[#] GAP [ ] tested'),
caption.replace('[ ] GAP', '[ ] tested'), caption.replace('[ ] GAP', '[#] GAP'),
caption + ' except refunds', caption + '\nLegend: [#] untested [ ] covered',
caption + '\nLegend:[#] untested [ ] covered',
caption + '\nsrc/billing.ts — coverage map[#] untested [ ] covered',
caption + '\nThis legend is withdrawn.', caption + '\nThis legend applies only if approved.',
]) {
const s = synthetic(); s.result.output = displayLegend(legend); expect(verdict(s).diagram).toBe(false);
}
const valid = displayLegend(caption);
for (const output of [
'```text\n' + caption + '\n```\n' + displayLegend(''),
valid.replace('processPayment', 'otherPayment'), valid.replace('refundPayment', 'otherRefund'),
valid.replace('return success [#]', 'return success not [#]'),
valid.replace('return success [#]', 'return success [#] -> [ ]'),
valid.replace('return refunded [ ]', 'return refunded [ ] -> [#]'),
valid.replace('return refunded [ ]', 'return refunded [ ] [#]'),
valid.replace('return success [#]', 'return success ├─ [#]'),
'````markdown\n' + valid + '\n````', 'Example:\n' + valid,
displayLegend('Legend: [✓] tested [✗] GAP ── covered', '✓', '✗'),
displayLegend('Legend: [✓] tested [✗] GAP ── branch except refunds', '✓', '✗'),
]) {
const s = synthetic(); s.result.output = output; expect(verdict(s).diagram).toBe(false);
}
const s = synthetic(); s.result.output = valid; s.result.transcript = [];
expect(verdict(s).diagram).toBe(true); expect(verdict(s).passed).toBe(false);
});
test('CI symbol legends remain current, unambiguous and owned by their diagram', () => {
for (const row of ciDiagrams.diagrams) {
const text = row.text, key = text.split('\n').find(line => line.startsWith('Legend:'))!;
for (const replacement of ['', '> ' + key, 'Source: ' + key, key + ' except refunds',
key.replace(/covered(?: by a test)?/, 'untested'),
key + '\nLegend: [✓] GAP [✗] covered', key + '\n [✓] GAP [✗] covered',
...['Sample:', 'Example legend:', 'Illustration:'].map(label => label + '\n' + key)]) {
const s = synthetic(); s.result.output = text.replace(key, replacement);
expect(verdict(s).diagram, replacement).toBe(false);
}
for (const status of ['This legend is withdrawn.', 'Assessment complete; This legend is `no longer current`.',
'**This legend** is “rejected”.', 'This legend applies only if approved.']) {
const s = synthetic(); s.result.output = text.replace(/\n```$/, '\n' + status + '\n```');
expect(verdict(s).diagram, status).toBe(false);
}
for (const output of ['Example:\n' + text, '````markdown\n' + text + '\n````',
text.replace(/^```[^\n]*/, '```json'), '```\n' + key + '\n```\n' + text.replace(key, '')]) {
const s = synthetic(); s.result.output = output; expect(verdict(s).diagram, output).toBe(false);
}
const s = synthetic(); s.result.output = text.replace(/\n```$/, '\nEarlier reviewer said "This legend is withdrawn."\n```');
expect(verdict(s).diagram).toBe(true);
}
});
test('six-column annotations cannot borrow sibling, prose or parallel-column markers', () => {
const text = '```\nLegend: [✓] covered by a test [✗] GAP — no test exercises this path\n'
+ 'processPayment(amount, currency)\n└── happy return success\n [✓] covered\n'
+ 'refundPayment(paymentId, reason)\n└── return refunded\n [✗] GAP\n```';
for (const output of [text.replace(' [✓]', 'unrelatedPayment()\n [✓]'),
text.replace(' [✓]', ' Earlier example:\n [✓]'),
text.replace(' [✓]', ' [✓]'),
text.replace(' [✓]', ' [✗]'), text.replace(' [✗]', ' [✓]'),
text.replace('└── happy return success\n [✓]', '└── happy return success ├── [✓]')]) {
const s = synthetic(); s.result.output = output; expect(verdict(s).diagram).toBe(false);
}
const s = synthetic(); s.result.output = text; expect(verdict(s).passed).toBe(true);
});
test.each([
'```text\nsrc/billing.ts\n├── refundPayment [UNTESTED]\n└── processPayment: happy path [TESTED]\n```',
'src/billing.ts\n├── processPayment: happy path [TESTED]\n└── refundPayment [UNTESTED]',
])('function order and optional fencing do not change valid coverage evidence: %s', output=>{
const s=synthetic();s.result.output=output;expect(verdict(s).diagram).toBe(true);
});
test('direct Read, literal cat/sed and delivered native line gutters are valid',()=>{
for(const command of ['cat -n src/billing.ts',"sed -n '1,200p' 'src/billing.ts'",'cat -- "src/billing.ts"']){
const s=synthetic();Object.assign(block(s,1),{name:'Bash',input:{command}});
block(s,2).content=fixture.files.source.split('\n').map((line,i)=>`${i+1}\t${line}`).join('\n');
expect(verdict(s).passed).toBe(true);
}
const s=synthetic();block(s,2).content=[{type:'text',text:fixture.files.source.split('\n').map((line,i)=>`${i+1}→${line}`).join('\n')}];
expect(verdict(s).passed).toBe(true);
});
function mixedDisplay(context: boolean) {
const s = synthetic();
const command = context
? 'cat review/specialists/testing.md && echo ==== SRC ==== && cat -n src/billing.ts && echo ==== TEST ==== && cat -n test/billing.test.ts && echo ==== GIT ==== && git log --oneline main..HEAD; git diff main --stat'
: 'cat -n test/billing.test.ts && git log --oneline main..feature/billing 2>/dev/null; git diff main...feature/billing --stat 2>/dev/null';
const numbered = (body: string) => body.replace(/\n$/, '').split('\n').map((line, index) => `${index + 1}\t${line}`).join('\n');
if (context) s.result.transcript.splice(1, 2);
const use = s.result.transcript.at(-2).message.content[0];
const result = s.result.transcript.at(-1).message.content[0];
Object.assign(use, {name: 'Bash', input: {command}});
result.content = context
? '# Testing Specialist Review Checklist\n\nCoverage Gaps\n==== SRC ====\n' + numbered(s.files.source.content)
+ '\n==== TEST ====\n' + numbered(s.files.tests.content) + '\n==== GIT ===='
: numbered(s.files.tests.content);
return {s, use, result};
}
test('mixed Git display tails retain separately delivered files and numbered reads after context', () => {
// Shell forms from the two failed 2026-09-20 paid /review captures.
for (const context of [false, true]) expect(verdict(mixedDisplay(context).s).passed).toBe(true);
});
test('mixed display reads retain ordered bodies and successful parent ownership', () => {
for (const context of [false, true]) for (const mutate of [
(x: ReturnType<typeof mixedDisplay>) => { x.result.is_error = true; },
(x: ReturnType<typeof mixedDisplay>) => { x.result.content = 'test/billing.test.ts was read'; },
(x: ReturnType<typeof mixedDisplay>) => { x.result.content = x.result.content.replace(/.*import \{ describe.*\n/, ''); },
(x: ReturnType<typeof mixedDisplay>) => { x.s.result.transcript.at(-1).session_id = 'foreign'; },
(x: ReturnType<typeof mixedDisplay>) => { x.s.result.transcript.at(-1).parent_tool_use_id = 'child'; },
(x: ReturnType<typeof mixedDisplay>) => { x.result.tool_use_id = 'unpaired'; },
(x: ReturnType<typeof mixedDisplay>) => { x.s.result.transcript.push(clone(x.s.result.transcript.at(-1))); },
]) {
const x = mixedDisplay(context); mutate(x); expect(verdict(x.s).testsRead).toBe(false);
}
for (const context of [false, true]) for (const suffix of [
'git diff main --output=src/billing.ts --stat', 'git diff main --ext-diff --stat',
'git diff main --stat > output.txt', 'git diff main --stat || echo ok',
]) {
const x = mixedDisplay(context); x.use.input.command = x.use.input.command.replace(/git diff[^;]+$/, suffix);
expect(verdict(x.s).testsRead).toBe(false);
}
for (const prefix of ['cat ../foreign.md', 'cat --help.md', 'cat /foreign.md', 'cat "$CONTEXT"', 'cat review/specialists/testing.md | head -2',
'false', 'python3 -c "pass"', 'echo -e "replacement"', 'cat review/specialists/testing.md; false']) {
const x = mixedDisplay(true); x.use.input.command = x.use.input.command.replace('cat review/specialists/testing.md', prefix);
expect(verdict(x.s).sourceRead).toBe(false); expect(verdict(x.s).testsRead).toBe(false);
}
const x = mixedDisplay(true); x.result.content = x.result.content.replace('==== SRC ====', '==== OTHER ====');
expect(verdict(x.s).sourceRead).toBe(false); expect(verdict(x.s).testsRead).toBe(false);
const repeated = mixedDisplay(true); repeated.result.content += '\n==== SRC ====';
expect(verdict(repeated.s).sourceRead).toBe(false); expect(verdict(repeated.s).testsRead).toBe(false);
const missing = mixedDisplay(true); missing.result.content = missing.result.content.slice(missing.result.content.indexOf('==== SRC ===='));
expect(verdict(missing.s).sourceRead).toBe(false); expect(verdict(missing.s).testsRead).toBe(false);
});
function boundDisplay(kind: 'and-log' | 'quoted-grep') {
const s = synthetic();
const numbered = (body: string) => body.replace(/\n$/, '').split('\n').map((line, index) => `${index + 1}\t${line}`).join('\n');
// Exact commands from the two completed, failed 2026-09-20 bound reruns.
const command = kind === 'and-log'
? 'cat -n src/billing.ts && echo ==== && cat -n test/billing.test.ts && echo ==== && git log --oneline main..HEAD && git diff main --stat'
: "grep -n -i 'diagram\\|coverage\\|tested\\|gap' review/SKILL.md | head -60; echo ======SRC; cat -n src/billing.ts; echo ======TEST; cat -n test/billing.test.ts; echo ======DIFF; git diff main...HEAD --stat";
s.result.transcript.splice(3, 2);
const use = block(s, 1), result = block(s, 2);
Object.assign(use, {name: 'Bash', input: {command}});
result.content = kind === 'and-log'
? numbered(s.files.source.content) + '\n====\n' + numbered(s.files.tests.content) + '\n===='
: '119: Test coverage gaps for stated requirements\n======SRC\n' + numbered(s.files.source.content)
+ '\n======TEST\n' + numbered(s.files.tests.content) + '\n======DIFF';
return {s, use, result};
}
test.each(['and-log', 'quoted-grep'] as const)('complete parent reads survive closed neighboring displays: %s', kind => {
expect(verdict(boundDisplay(kind).s)).toEqual({sourceRead:true, testsRead:true, diagram:true, passed:true, failures:[]});
});
test('neighboring log and quoted grep displays cannot replace complete owned delivery', () => {
for (const kind of ['and-log', 'quoted-grep'] as const) for (const mutate of [
(x: ReturnType<typeof boundDisplay>) => { x.result.is_error = true; },
(x: ReturnType<typeof boundDisplay>) => { x.result.content = 'src/billing.ts and test/billing.test.ts were read'; },
(x: ReturnType<typeof boundDisplay>) => { x.s.result.transcript[2].session_id = 'foreign'; },
(x: ReturnType<typeof boundDisplay>) => { x.s.result.transcript[2].parent_tool_use_id = 'child'; },
(x: ReturnType<typeof boundDisplay>) => { x.s.result.transcript[1].parent_tool_use_id = 'child'; },
(x: ReturnType<typeof boundDisplay>) => { x.result.tool_use_id = 'unpaired'; },
(x: ReturnType<typeof boundDisplay>) => { x.s.result.transcript.push(clone(x.s.result.transcript[2])); },
]) {
const x = boundDisplay(kind); mutate(x);
expect(verdict(x.s).sourceRead).toBe(false); expect(verdict(x.s).testsRead).toBe(false);
}
for (const kind of ['and-log', 'quoted-grep'] as const) for (const [key, line] of [
['sourceRead', /.*export function processPayment.*\n/], ['testsRead', /.*import \{ describe.*\n/],
] as const) {
const x = boundDisplay(kind); x.result.content = x.result.content.replace(line, '');
expect(verdict(x.s)[key]).toBe(false); expect(verdict(x.s).passed).toBe(false);
}
});
test('closed neighboring log and grep grammars reject unsafe lookalikes', () => {
for (const display of [
'git log --oneline main..HEAD --output=src/billing.ts', 'git log --oneline main..HEAD --format=%B',
'git log --oneline main..HEAD --ext-diff', 'git log --oneline main..HEAD > output.txt',
'git log --oneline "main..HEAD"', 'git log --oneline main..HEAD || echo ok',
]) {
const x = boundDisplay('and-log'); x.use.input.command = x.use.input.command.replace('git log --oneline main..HEAD', display);
expect(verdict(x.s).sourceRead).toBe(false); expect(verdict(x.s).testsRead).toBe(false);
}
for (const display of [
'grep -n -i "$(touch sentinel)" review/SKILL.md | head -60',
'grep -n -i "`touch sentinel`" review/SKILL.md | head -60',
"grep -n -i 'diagram\\|coverage' --help | head -60",
"grep -n -i 'diagram\\|coverage' review/SKILL.md > output.txt",
"grep -n -i 'diagram\\|coverage' review/SKILL.md | python3 -c 'pass'",
"grep -n -i 'diagram\\ncoverage' review/SKILL.md | head -60",
]) {
const x = boundDisplay('quoted-grep'); x.use.input.command = x.use.input.command.replace(/^[^;]+/, display);
expect(verdict(x.s).sourceRead).toBe(false); expect(verdict(x.s).testsRead).toBe(false);
}
});
test('each exact source and test file must be successfully delivered',()=>{
for(const mutate of [
(s:any)=>{block(s,2).content='src/billing.ts was read';},
(s:any)=>{block(s,2).content=fixture.files.source.split('\n').slice(0,3).join('\n');},
(s:any)=>{block(s,2).is_error=true;},
(s:any)=>{block(s,3).input.file_path=s.files.source.path;},
(s:any)=>{block(s,4).content=fixture.files.source;},
(s:any)=>{block(s,1).input.file_path=path.join(s.files.cwd,'other/billing.ts');},
(s:any)=>{s.files.tests.path=s.files.source.path;},
]){const s=synthetic();mutate(s);expect(verdict(s).passed).toBe(false);}
});
test('unpaired, repeated, child and foreign events cannot supply parent file evidence',()=>{
for(const mutate of [
(s:any)=>{s.result.transcript.splice(1,1);},
(s:any)=>{[s.result.transcript[1],s.result.transcript[2]]=[s.result.transcript[2],s.result.transcript[1]];},
(s:any)=>{s.result.transcript[2].session_id='foreign';},
(s:any)=>{s.result.transcript[2].parent_tool_use_id='agent';},
(s:any)=>{s.result.transcript[1].parent_tool_use_id='agent';},
(s:any)=>{s.result.transcript[2].message.role='assistant';},
(s:any)=>{s.result.transcript.push(clone(s.result.transcript[2]));},
(s:any)=>{s.result.transcript.push(clone(s.result.transcript[1]));},
(s:any)=>{s.result.transcript[0].cwd+='/sibling';},
(s:any)=>{s.result.transcript.push(clone(s.result.transcript[0]));},
(s:any)=>{s.result.transcript[0].session_id='foreign';},
(s:any)=>{s.result.transcript[0].type='user';},
(s:any)=>{s.result.transcript.shift();},
]){const s=synthetic();mutate(s);expect(verdict(s).passed).toBe(false);}
});
test('quoted metadata, counters and undeclared shell reads do not substitute for actual delivery',()=>{
for(const command of [
"echo 'cat src/billing.ts'",'false && cat src/billing.ts','cat src/billing.ts | head -2',
'cd ../sibling; cat src/billing.ts',"if true; then cat src/billing.ts; fi",'cat "$SOURCE"',
"cat <<'EOF'\ncat src/billing.ts\nEOF",'f() {\ncat src/billing.ts\n}',
'(\ncat src/billing.ts\n)',
]){const s=synthetic();Object.assign(block(s,1),{name:'Bash',input:{command}});expect(verdict(s).passed).toBe(false);}
const s=synthetic();s.result.toolCalls=[{tool:'Read',input:{file_path:s.files.source.path}},{tool:'Read',input:{file_path:s.files.tests.path}}];
s.result.transcript=[s.result.transcript[0],{type:'assistant',session_id:'parent',message:{role:'assistant',content:[{type:'text',text:JSON.stringify(s.result.transcript.slice(1))}]}}];
expect(verdict(s).sourceRead).toBe(false);expect(verdict(s).testsRead).toBe(false);
});
test('a commented read cannot borrow printed bytes; quoted hash paths remain literal',()=>{
const s=synthetic();
const command=`printf '${Buffer.from(fixture.files.source).toString('base64')}' | base64 -d; # only printed bytes; cat src/billing.ts`;
Object.assign(block(s,1),{name:'Bash',input:{command}});
expect(verdict(s).sourceRead).toBe(false);
const quoted=synthetic();quoted.files.source.path=path.join(quoted.files.cwd,'src/billing#branch.ts');
Object.assign(block(quoted,1),{name:'Bash',input:{command:"cat 'src/billing#branch.ts'"}});
expect(verdict(quoted).sourceRead).toBe(true);
});
test('completion and tool errors remain final gate failures despite genuine delivery',()=>{
for(const exitReason of ['timeout','exit_code_1']){const s=synthetic();s.result.exitReason=exitReason;expect(verdict(s).passed).toBe(false);}
const s=synthetic();s.result.browseErrors=['read failed'];expect(verdict(s).passed).toBe(false);
});
test('coverage markers must belong to the seeded payment and refund functions',()=>{
for(const output of [
diagram.replace('[TESTED]','[UNTESTED]'),diagram.replace('[UNTESTED]','[TESTED]'),
diagram.replace('processPayment','processPaymentExample'),diagram.replace('refundPayment','refundPaymentExample'),
'```\n├── processPayment: happy path [TESTED]\n└── refundPayment [TESTED]\n└── unrelatedPayment [UNTESTED]\n```',
'```\n├── processPayment: happy path [TESTED]\n```\n```\n└── refundPayment [UNTESTED]\n```',
]){const s=synthetic();s.result.output=output;expect(verdict(s).diagram).toBe(false);}
});
test('quoted and nested source examples are not the generated coverage diagram',()=>{
for(const output of [diagram.split('\n').map(line=>'> '+line).join('\n'), '````markdown\n'+diagram+'\n````']){
const s=synthetic();s.result.output=output;expect(verdict(s).diagram).toBe(false);
}
});
test.each([
diagram.replace('[UNTESTED]','[NOT UNTESTED]'),
diagram.replace('[UNTESTED]','[UNTESTED] is false; this function is fully covered.'),
'Example only; this diagram is not the audit result.\n'+diagram,
])('negated gaps and explicitly labeled examples are not audit findings: %s', output=>{
const s=synthetic();s.result.output=output;expect(verdict(s).diagram).toBe(false);
});
test('collector receives exactly the asserted verdict even when process exit succeeded',()=>{
for(const valid of [true,false]){
const s=synthetic();if(!valid)s.result.output='No diagram produced.';
Object.assign(s.result,{toolCalls:[],duration:1,costEstimate:{estimatedCost:0,turnsUsed:1,estimatedTokens:1}});
const v=verdict(s),entries:any[]=[];
recordE2E({addTest:(entry:any)=>entries.push(entry)} as any,'coverage','fixture',s.result,{passed:v.passed,error:v.failures.length?v.failures.join('; '):undefined});
expect(entries).toHaveLength(1);expect(entries[0].passed).toBe(valid);expect(entries[0].error).toBe(valid?undefined:v.failures.join('; '));
}
});
});
describe('coverage-audit-af', () => {
const fixture = fixture_coverage_audit_af;
const actual = (index: number) => structuredClone(fixture.rows[index]!);
const files = (row: typeof fixture.rows[number]) => ({cwd:row.cwd,
source:{path:`${row.cwd}/src/billing.ts`,content:fixture.files.source},
tests:{path:`${row.cwd}/test/billing.test.ts`,content:fixture.files.tests}});
for (let i=0;i<fixture.rows.length;i++) test(`AF exact public coverage audit ${i+1} binds delivered files and seeded diagram`, () => {
const row=actual(i); expect(coverageAuditVerdict(row.result, files(row))).toEqual({sourceRead:true,testsRead:true,diagram:true,passed:true,failures:[]});
});
function delivered(command: string, mutate?: (events: any[]) => void) {
const row=actual(2), session=row.sessionId;
const transcript:any[]=[
{type:'system',subtype:'init',session_id:session,cwd:row.cwd},
{type:'assistant',session_id:session,parent_tool_use_id:null,message:{role:'assistant',content:[{type:'tool_use',id:'read-pair',name:'Bash',input:{command}}]}},
{type:'user',session_id:session,parent_tool_use_id:null,message:{role:'user',content:[{type:'tool_result',tool_use_id:'read-pair',is_error:false,content:fixture.files.source+'\n----\n'+fixture.files.tests}]}},
];
mutate?.(transcript);
return coverageAuditVerdict({...row.result,transcript},files(row));
}
const both = 'cat -n src/billing.ts && cat -n test/billing.test.ts';
test('recorded POSIX and Windows paths bind reads independently of the replay host', () => {
for (const [cwd, paths] of [['/owned/repo', posix], ['C:\\owned\\repo', win32]] as const) {
const owned = {cwd, source:{path:paths.join(cwd,'src/billing.ts'),content:fixture.files.source},
tests:{path:paths.join(cwd,'test/billing.test.ts'),content:fixture.files.tests}};
const transcript = [
{type:'system',subtype:'init',session_id:'owned',cwd},
{type:'assistant',session_id:'owned',message:{role:'assistant',content:[{type:'tool_use',id:'pair',name:'Bash',input:{command:both}}]}},
{type:'user',session_id:'owned',message:{role:'user',content:[{type:'tool_result',tool_use_id:'pair',is_error:false,content:fixture.files.source+'\n'+fixture.files.tests}]}},
];
expect(coverageAuditReadEvidence(transcript,owned)).toEqual({sourceRead:true,testsRead:true});
expect(coverageAuditReadEvidence(transcript,{...owned,source:{...owned.source,path:paths.join(cwd,'../foreign.ts')}}))
.toEqual({sourceRead:false,testsRead:false});
expect(coverageAuditReadEvidence(transcript,{...owned,source:{...owned.source,path:cwd+paths.sep+'src'+paths.sep+'..'+paths.sep+'src'+paths.sep+'billing.ts'}}))
.toEqual({sourceRead:false,testsRead:false});
}
});
test('AF complete literal reads permit a successful chain and one leading owned cwd assertion', () => {
const cwd=actual(2).cwd;
for (const command of [both, `cd ${cwd}; cat -n src/billing.ts; cat -n test/billing.test.ts`, `cd '${cwd}' && ${both}`, 'cat -n src/billing.ts; echo ----; cat -n test/billing.test.ts']) {
const v=delivered(command); expect(v.sourceRead).toBe(true); expect(v.testsRead).toBe(true); expect(v.passed).toBe(true);
}
});
test('AF the new conditional-chain grammar conservatively rejects mixed separators', () => {
const v=delivered(`cd ${actual(2).cwd}; ${both}`);
expect(v.sourceRead).toBe(false); expect(v.testsRead).toBe(false);
});
test('AF read recognition rejects foreign or midstream cwd changes and nonliteral targets', () => {
const cwd=actual(2).cwd;
for (const command of [`cd /foreign; ${both}`, `cat -n src/billing.ts; cd /foreign; cat -n test/billing.test.ts`,
`cat -n src/billing.ts; cd ${cwd}; cat -n test/billing.test.ts`, `cd "$PWD"; ${both}`, `cd ${cwd}/..; ${both}`]) {
const v=delivered(command); expect(v.sourceRead).toBe(false); expect(v.testsRead).toBe(false);
}
});
test('AF a printed, conditional or skipped read cannot borrow delivered-looking file contents', () => {
for (const command of [`false && ${both}`, `if false; then ${both}; fi`, `echo '${both}'`, `exit; ${both}`,
`# ${both}`, `cat <<'EOF'\n${both}\nEOF`, `(${both})`, `f() { ${both}; }`, `printf '%s' '${both}'`,
`printf expected; false && ${both}; true`, `${both} > result.txt`]) {
const v=delivered(command); expect(v.sourceRead).toBe(false); expect(v.testsRead).toBe(false);
}
});
test('AF an owned cwd does not authorize mutations or interpreters around a read', () => {
for (const neighbor of ['rm -f src/billing.ts', 'python3 -c "pass"', 'echo fake > src/billing.ts',
'grep data backup.txt | tee src/billing.ts', 'git diff --output=src/billing.ts',
"git diff '--output=src/billing.ts'", "git diff --output'='src/billing.ts",
'git diff --out=src/billing.ts', 'git diff --ext-diff']) {
const result = delivered(`cd ${actual(2).cwd}; ${neighbor}; cat src/billing.ts; cat test/billing.test.ts`);
expect(result.sourceRead).toBe(false); expect(result.testsRead).toBe(false);
}
});
test('AF added command forms retain exact parent request/result success and delivered-content binding', () => {
const mutations:Array<(events:any[])=>void>=[
e=>{e[0].cwd='/foreign';}, e=>{e[2].session_id='foreign';},
e=>{e[1].parent_tool_use_id='child';}, e=>{e[2].message.content[0].is_error=true;},
e=>{e[2].message.content[0].tool_use_id='unpaired';},
e=>{e[2].message.content[0].content='The two filenames were read.';},
];
for (const mutate of mutations) { const v=delivered(both,mutate); expect(v.sourceRead).toBe(false); expect(v.testsRead).toBe(false); }
const onlySource=delivered(both,e=>{e[2].message.content[0].content=fixture.files.source;});
expect(onlySource.sourceRead).toBe(true); expect(onlySource.testsRead).toBe(false);
});
const flat = (legend = 'Legend: [✓] tested [✗] GAP') => `\`\`\`text\n${legend}\nprocessPayment(amount, currency)\n├── [✓] happy path USD\nrefundPayment(paymentId, reason)\n└── [✗] happy path refund\n\`\`\``;
function diagram(output:string) { const row=actual(0);return coverageAuditVerdict({...row.result,output},files(row)).diagram; }
test('AF flat function roots and same-block legend symbols preserve seeded coverage ownership', () => {
expect(diagram(flat())).toBe(true);
expect(diagram(flat().replaceAll('✓','✔').replaceAll('✗','✘'))).toBe(true);
expect(diagram(flat().replace('processPayment(amount, currency)\n├── [✓] happy path USD\nrefundPayment(paymentId, reason)\n└── [✗] happy path refund',
'refundPayment(paymentId, reason)\n├── [✗] happy path refund\nprocessPayment(amount, currency)\n└── [✓] happy path USD'))).toBe(true);
});
test('AF symbol-only markers need an unambiguous legend in their own diagram block', () => {
for (const output of [flat(''),flat('Legend: [✓] GAP [✗] tested'),flat('Legend: [✓] tested [✗] tested'),
flat('Legend: [✓] tested [✗] GAP [✗] tested'),
`\`\`\`text\nLegend: [✓] tested [✗] GAP\n\`\`\`\n${flat('')}`]) expect(diagram(output)).toBe(false);
});
test('AF flat roots cannot borrow another function subtree or a quoted/example diagram', () => {
for (const output of [
flat().replace('├── [✓] happy path USD','├── untested amount guard\nunrelatedHelper()\n└── [✓] happy path USD'),
flat().replace('refundPayment(paymentId, reason)','unrelatedRefund(paymentId, reason)'),
flat().replace('processPayment(amount, currency)','processPaymentOther(amount, currency)'),
flat().split('\n').map(line=>'> '+line).join('\n'),
'````markdown\n'+flat()+'\n````',
flat().replace('Legend:','Example diagram:\nLegend:'),
]) expect(diagram(output)).toBe(false);
});
test('AF a legend cannot override an explicitly negated marker on its own branch', () => {
for (const output of [
flat().replace('[✗] happy path refund','not [✗] happy path refund'),
flat().replace('[✗] happy path refund','[✗] is false; this branch is tested'),
flat().replace('[✓] happy path USD','not [✓] happy path USD'),
]) expect(diagram(output)).toBe(false);
});
test('AF symbol gaps retain affirmative legend ownership and reject same-branch contradictions', () => {
for (const output of [
flat().replace('[✗] happy path refund', '[✗] happy path refund (marker is incorrect; this branch is fully tested)'),
flat().replace('[✗] happy path refund', '[✗] happy path refund — no coverage gap exists'),
flat('An unproven hypothesis: [✓] tested [✗] GAP'),
flat("The source says '[✓] tested [✗] GAP'"),
]) expect(diagram(output)).toBe(false);
expect(diagram(flat())).toBe(true);
expect(diagram(flat('[✓] tested [✗] GAP'))).toBe(true);
expect(diagram(flat('src/billing.ts — test coverage map [✓] tested [✗] GAP'))).toBe(true);
});
});
describe('coverage-audit-aw', () => {
const fixture = fixture_coverage_audit_aw;
const fresh=(n=0)=>{
const r=structuredClone(fixture.reads[n]!),files={cwd:r.cwd,source:{path:r.cwd+'/src/billing.ts',content:fixture.source},tests:{path:r.cwd+'/test/billing.test.ts',content:fixture.tests}};
const transcript:any[]=[{type:'system',subtype:'init',session_id:r.sessionId,cwd:r.cwd},{type:'assistant',session_id:r.sessionId,parent_tool_use_id:null,message:{role:'assistant',content:[{type:'tool_use',id:r.toolUseId,name:'Bash',input:{command:r.command}}]}},{type:'user',session_id:r.sessionId,parent_tool_use_id:null,message:{role:'user',content:[{type:'tool_result',tool_use_id:r.toolUseId,is_error:false,content:r.outputExcerpt}]}}];
return {files,transcript};
};
const reads=(x:ReturnType<typeof fresh>)=>coverageAuditReadEvidence(x.transcript,x.files);
const diagram=(text:string)=>{const x=fresh();return coverageAuditVerdict({exitReason:'success',browseErrors:[],output:text,transcript:x.transcript} as any,x.files).diagram;};
describe('Coverage audit owned display composition and marker continuations',()=>{
test.each([0,1,2,3])('credits exact complete public file delivery %i',n=>{
expect(fixture.provenance.actualCollectorFailuresRetained).toBe(true);expect(reads(fresh(n))).toEqual({sourceRead:true,testsRead:true});
});
test.each(['failed result','foreign session','foreign cwd','foreign tool id','sidechain','missing result','repeated result','partial body','forged body'])('rejects %s',form=>{
for(let n=0;n<4;n++){const x=fresh(n),e=x.transcript[2],b=e.message.content[0];if(form==='failed result')b.is_error=true;else if(form==='foreign session')e.session_id='foreign';else if(form==='foreign cwd')x.transcript[0].cwd+='/other';else if(form==='foreign tool id')b.tool_use_id='foreign';else if(form==='sidechain')e.parent_tool_use_id='parent';else if(form==='missing result')x.transcript.pop();else if(form==='repeated result')x.transcript.push(structuredClone(e));else if(form==='partial body')b.content=b.content.replace(/.*(?:export function processPayment|import \{ describe).*\n/g,'');else b.content='src/billing.ts and test/billing.test.ts were read';expect(reads(x)).toEqual({sourceRead:false,testsRead:false});}
});
test.each(['foreign paths','printf forgery','echo escape forgery','expansion','double quoted expansion','awk execution','changed ordered prefix'])('rejects unsupported or unowned command: %s',form=>{
const n=form==='awk execution'?2:form==='changed ordered prefix'?3:0,x=fresh(n),u=x.transcript[1].message.content[0];u.input.command=form==='foreign paths'?u.input.command.replaceAll('src/billing.ts','other/billing.ts').replaceAll('test/billing.test.ts','other/billing.test.ts'):form==='printf forgery'?"printf 'fixture body'":form==='echo escape forgery'?"echo -e 'fake\\nbody'":form==='expansion'?u.input.command+'; echo $(cat source)':form==='double quoted expansion'?u.input.command+'; echo "$HOME"':form==='awk execution'?u.input.command.replace('{f=1}','{system("cat forged") }'):u.input.command.replace('=== src/billing.ts ===','=== other.ts ===');expect(reads(x)).toEqual({sourceRead:false,testsRead:false});
});
test.each([0,1])('accepts the exact public current diagram %i',n=>expect(diagram(fixture.diagrams[n]!.text)).toBe(true));
test.each(['missing key','inverted checkbox key','withdrawn key','foreign function','quoted source','not covered','not missing'])('rejects contradictory or unowned checkbox coverage: %s',form=>{
const text=fixture.diagrams[0]!.text;const changed=form==='missing key'?text.replace(/^Legend:.*\n/m,''):form==='inverted checkbox key'?text.replace('[x] tested [ ] GAP','[x] untested [ ] tested'):form==='withdrawn key'?text.replace('src/billing.ts\n│','This legend is withdrawn.\nsrc/billing.ts\n│'):form==='foreign function'?text.replaceAll('refundPayment','otherPayment'):form==='quoted source'?'Example only:\n'+text:form==='not covered'?text.replace("[x] 'processes valid payment'","[ ] GAP"):text.replaceAll('[ ] GAP','[x] tested');expect(diagram(changed)).toBe(false);
});
test('continuations keep their own row and cannot borrow from prose or a distant column',()=>{
const text=fixture.diagrams[1]!.text;
expect(diagram(text.replace('│ [✓] billing.test.ts:6', '│ Earlier example:\n│ [✓] billing.test.ts:6'))).toBe(false);
expect(diagram(text.replace('│ [✓] billing.test.ts:6', ' [✓] billing.test.ts:6'))).toBe(false);
expect(diagram(text.replace('│ [✓] billing.test.ts:6', '│ [✗] billing.test.ts:6'))).toBe(false);
expect(diagram(text.replaceAll('[✗] GAP','[✓] tested').replace('[✗] untested (GAP)','[✗] untested (GAP)'))).toBe(false);
});
});
});
describe('coverage-audit-shell-legend-at', () => {
const captured = captured_coverage_audit_shell_legend_at;
const both = { sourceRead: true, testsRead: true };
const neither = { sourceRead: false, testsRead: false };
function owned(index: number) {
const row = structuredClone(captured[index]!) as any;
const useEvent = row.result.transcript.find((event: any) => event.message?.content.some((block: any) =>
block.type === 'tool_use' && block.name === 'Bash' && block.input.command.includes('cat -n src/billing.ts')));
const use = useEvent.message.content.find((block: any) => block.type === 'tool_use' && block.name === 'Bash' && block.input.command.includes('cat -n src/billing.ts'));
const resultEvent = row.result.transcript.find((event: any) => event.message?.content.some((block: any) => block.type === 'tool_result' && block.tool_use_id === use.id));
row.result.transcript = [row.result.transcript.find((event: any) => event.type === 'system' && event.subtype === 'init'), useEvent, resultEvent];
return { row, use, resultEvent, delivered: resultEvent.message.content.find((block: any) => block.tool_use_id === use.id) };
}
function reads(index: number, mutate?: (s: ReturnType<typeof owned>) => void) {
const s = owned(index); mutate?.(s);
return coverageAuditReadEvidence(s.row.result.transcript, s.row.files);
}
const flat = (legend = 'Legend [ OK ] covered [ GAP ] no test') =>
'```text\nprocessPayment(amount, currency)\n├── happy path return success [ OK ]\nrefundPayment(paymentId, reason)\n└── happy path return refunded [ GAP ]\n' + legend + '\n```';
const diagram = (output: string) => coverageAuditVerdict({ ...captured[1]!.result, output } as any, captured[1]!.files).diagram;
test('exact public AT first, retry and engineering outputs retain all required native evidence', () => {
expect(captured.map(row => row.recordedPassed)).toEqual([false, false, true]);
for (const row of captured) expect(coverageAuditVerdict(row.result as any, row.files)).toEqual({ ...both, diagram: true, passed: true, failures: [] });
expect(reads(0)).toEqual(both); expect(reads(1)).toEqual(both);
});
test('literal grep display options and filename captions do not own source bytes', () => {
for (const flags of ['-n', '-n -i', '-n -B1 -A200', '-n -i -B3 -A40']) {
expect(reads(0, s => { s.use.input.command = s.use.input.command.replace('-n -i -B3 -A40', flags); })).toEqual(both);
}
for (const replace of ['echo \'=== another-file.md ===\'', 'echo "--- src/billing.ts ---"', 'echo']) {
expect(reads(1, s => { s.use.input.command = s.use.input.command.replace('echo "=== testing.md ==="', replace); })).toEqual(both);
}
expect(reads(1, s => { s.use.input.command = s.use.input.command.replace('git diff main --stat', 'git diff HEAD~1 --stat'); })).toEqual(both);
});
test('escaped grep patterns keep a closed flag and literal operand grammar', () => {
for (const replacement of ['-n -i -B3 -A40 -f other', '-n -i --include=*', '-n -B-1', '-n -A100000', '-n -i -B3 -A40; false']) {
expect(reads(0, s => { s.use.input.command = s.use.input.command.replace('-n -i -B3 -A40', replacement); })).toEqual(neither);
}
for (const operand of ['-f/tmp/foreign', '"-f/tmp/foreign"', 'review/SKILL.md --include=*']) {
expect(reads(0, s => { s.use.input.command = s.use.input.command.replace('review/SKILL.md |', operand + ' |'); })).toEqual(neither);
}
});
test('successful conditional display paths reject execution, substitutions and hidden failure', () => {
for (const replacement of [
'echo -e "=== testing.md ==="', 'printf "=== testing.md ==="', 'echo "$(cat fake)"', 'echo `cat fake`',
'echo "=== testing.md ==="; false', 'false || echo "=== testing.md ==="', 'unknown',
'echo "cat -n src/billing.ts"', 'echo "=== testing.md ===\\nreplacement"',
'cd ../sibling', 'env PATH=/tmp cat fake', 'echo "=== testing.md ===" > src/billing.ts',
]) expect(reads(1, s => { s.use.input.command = s.use.input.command.replace('echo "=== testing.md ==="', replacement); })).toEqual(neither);
for (const command of ['git diff --ext-diff --stat', 'git diff main --output=src/billing.ts --stat', 'git -c core.pager=evil diff main --stat', 'git diff --no-index main --stat', 'git diff main --stat || echo ok']) {
expect(reads(1, s => { s.use.input.command = s.use.input.command.replace('git diff main --stat', command); })).toEqual(neither);
}
});
test('a valid display path still requires one complete successful owned delivery', () => {
for (const index of [0, 1]) for (const mutate of [
(s: ReturnType<typeof owned>) => { s.delivered.is_error = true; },
(s: ReturnType<typeof owned>) => { s.delivered.content = 'src/billing.ts and test/billing.test.ts were read'; },
(s: ReturnType<typeof owned>) => { s.delivered.content = s.row.files.source.content.slice(0, 80); },
(s: ReturnType<typeof owned>) => { s.resultEvent.session_id = 'foreign'; },
(s: ReturnType<typeof owned>) => { s.resultEvent.parent_tool_use_id = 'child'; },
(s: ReturnType<typeof owned>) => { s.delivered.tool_use_id = 'foreign'; },
(s: ReturnType<typeof owned>) => { s.row.result.transcript.push(structuredClone(s.resultEvent)); },
(s: ReturnType<typeof owned>) => { s.use.input.command = s.use.input.command.replace('cat -n src/billing.ts', 'echo src/billing.ts').replace('cat -n test/billing.test.ts', 'echo test/billing.test.ts'); },
]) expect(reads(index, mutate)).toEqual(neither);
expect(reads(1, s => { s.delivered.content = s.row.files.source.content; })).toEqual({ sourceRead: true, testsRead: false });
expect(reads(1, s => { s.delivered.content = s.row.files.tests.content; })).toEqual({ sourceRead: false, testsRead: true });
});
test('text statuses use the declared local meanings with whitespace and either pair order', () => {
for (const legend of ['Legend [ OK ] covered [ GAP ] no test', 'Legend: [OK] tested | [GAP] untested', 'Legend: [ GAP ] no test; [ OK ] covered']) expect(diagram(flat(legend))).toBe(true);
expect(diagram(flat().replaceAll('[ OK ]', '[OK]').replaceAll('[ GAP ]', '[GAP]'))).toBe(true);
expect(diagram(flat().replace('Legend [ OK ] covered [ GAP ] no test\n', '').replace('processPayment', 'Legend [ OK ] covered [ GAP ] no test\nprocessPayment'))).toBe(true);
});
test('missing, malformed, contradictory or foreign text legends cannot grant coverage', () => {
for (const legend of ['', '> Legend [ OK ] covered [ GAP ] no test', '"Legend [ OK ] covered [ GAP ] no test"',
'Example: Legend [ OK ] covered [ GAP ] no test', 'If enabled, Legend [ OK ] covered [ GAP ] no test',
'Legend [ OK ] no test [ GAP ] covered', 'Legend [ OK ] covered [ GAP ] covered',
'Legend [ OK ] covered [ OK ] no test', 'Legend [ OK ] covered [ GAP ] no test except refunds',
'Legend [ OK ] covered [ GAP ] no test\nLegend [ OK ] no test [ GAP ] covered',
]) expect(diagram(flat(legend))).toBe(false);
expect(diagram('```text\nLegend [ OK ] covered [ GAP ] no test\n```\n' + flat(''))).toBe(false);
for (const status of ['cancelled', 'canceled', 'rejected', 'retracted', 'withdrawn', "'withdrawn'", '‘superseded’', '`no longer current`', '"not current"']) {
expect(diagram(flat('Legend [ OK ] covered [ GAP ] no test\nThis legend is ' + status + '.'))).toBe(false);
}
expect(diagram(flat('Legend [ OK ] covered [ GAP ] no test\n> An old note said: "This legend is withdrawn."'))).toBe(true);
});
test('text marker corrections grant only the final unambiguous owned row state', () => {
expect(diagram(flat().replace('success [ OK ]', 'success [ GAP ] -> [ OK ]'))).toBe(true);
expect(diagram(flat().replace('refunded [ GAP ]', 'refunded [ OK ] → [ GAP ]'))).toBe(true);
for (const [old, replacement] of [
['success [ OK ]', 'success COVERED [ OK ] → [ GAP ]'],
['refunded [ GAP ]', 'refunded UNTESTED [ GAP ] -> [ OK ]'],
['success [ OK ]', 'success COVERED [ OK ] [ GAP ]'],
['refunded [ GAP ]', 'refunded [GAP] [ GAP ] [ OK ]'],
['success [ OK ]', 'success not [ OK ]'], ['refunded [ GAP ]', 'refunded [ GAP ] is incorrect'],
['success [ OK ]', 'success not covered [ OK ]'], ['refunded [ GAP ]', 'refunded no coverage gaps [ GAP ]'],
]) expect(diagram(flat().replace(old!, replacement!))).toBe(false);
});
test('text coverage markers retain function, subtree, column and source ownership', () => {
for (const output of [
flat().replace('processPayment', 'otherPayment'), flat().replace('refundPayment', 'otherRefund'),
flat().replace('├── happy', 'otherFunction()\n├── happy'), flat().replace('└── happy', 'otherFunction()\n└── happy'),
flat().replace('success [ OK ]', 'success ├── [ OK ]'), flat().replace('refunded [ GAP ]', 'refunded └── [ GAP ]'),
flat().split('\n').map(line => '> ' + line).join('\n'), '````markdown\n' + flat() + '\n````', 'Example:\n' + flat(),
]) expect(diagram(output)).toBe(false);
});
});
describe('coverage-checkbox-tail-av', () => {
const fixture = fixture_coverage_checkbox_tail_av;
const both={sourceRead:true,testsRead:true}, neither={sourceRead:false,testsRead:false};
const fresh=(i=0)=>{const row=structuredClone(fixture.attempts[i]!) as any;return{row,use:row.result.transcript[1].message.content[0],ack:row.result.transcript[2].message.content[0]};};
const reads=(mutate:(s:ReturnType<typeof fresh>)=>void=()=>{})=>{const s=fresh();mutate(s);return coverageAuditReadEvidence(s.row.result.transcript,s.row.files);};
const base='```text\nprocessPayment(amount, currency)\n├── happy return success [x]\nrefundPayment(paymentId, reason)\n└── happy return refunded [ ]\nLegend: [x] tested [ ] no test\n```';
const diagram=(output:string)=>{const {row}=fresh();return coverageAuditVerdict({...row.result,output},row.files).diagram;};
test('both exact public attempts now provide their delivered files and owned checkbox diagram',()=>{
expect(fixture.provenance.paidOutcomesReclassified).toBe(false);
for(const row of fixture.attempts){expect(row.provenance.recordedPassed).toBe(false);expect(coverageAuditVerdict(row.result as any,row.files)).toEqual({...both,diagram:true,passed:true,failures:[]});}
});
test('mixed display tail accepts only the two ordered owned reads and literal separators',()=>{
expect(reads()).toEqual(both);
for(const revision of ['HEAD','HEAD~1','main..HEAD'])expect(reads(s=>{s.use.input.command=s.use.input.command.replace('main..HEAD',revision).replace('diff main','diff '+revision);})).toEqual(both);
expect(reads(s=>{s.use.input.command=s.use.input.command.replaceAll('echo ====','echo ----');s.ack.content=s.ack.content.replaceAll('====','----');})).toEqual(both);
expect(reads(s=>{s.use.input.command=s.use.input.command.replace('src/billing.ts',"'src/billing.ts'").replace('test/billing.test.ts','"test/billing.test.ts"');})).toEqual(both);
expect(reads(s=>{s.ack.content=[{type:'text',text:s.ack.content}];})).toEqual(both);
expect(reads(s=>{s.ack.content+='\nabc1234 harmless commit subject\n src/billing.ts | 2 ++\n 1 file changed, 2 insertions(+)';})).toEqual(both);
});
test.each([
'false && cat -n src/billing.ts && cat -n test/billing.test.ts && git log --oneline main..HEAD; git diff main --stat',
'cat -n src/billing.ts && false && cat -n test/billing.test.ts && git log --oneline main..HEAD; git diff main --stat',
'cat -n fake.ts && cat -n test/billing.test.ts && git log --oneline main..HEAD; git diff main --stat',
'cat -n src/billing.ts && cat -n src/billing.ts && git log --oneline main..HEAD; git diff main --stat',
'cat -n src/billing.ts && cat -n test/billing.test.ts; git log --oneline main..HEAD; git diff main --stat',
'cat -n src/billing.ts; cat -n test/billing.test.ts && git log --oneline main..HEAD; git diff main --stat',
])('a failed, skipped, duplicate or unrelated read cannot borrow display-tail success: %s',command=>{
expect(reads(s=>{s.use.input.command=command;})).toEqual(neither);
});
test.each([
['git log --oneline main..HEAD','git log --format=%B main..HEAD'],
['git log --oneline main..HEAD','git log --oneline --output=src/billing.ts main..HEAD'],
['git log --oneline main..HEAD','git -c core.pager=evil log --oneline main..HEAD'],
['git diff main --stat','git diff --ext-diff main --stat'],
['git diff main --stat','git diff --no-index main --stat'],
['git diff main --stat','git diff main --stat; echo extra'],
['git diff main --stat','git diff main --stat > src/billing.ts'],
['echo ====','echo replacement'],['echo ====','printf ===='],['echo ====','echo -e "\\nreplacement"'],
['cat -n src/billing.ts','cat -n src/billing.ts > test/billing.test.ts'],
['cat -n src/billing.ts','cat -n $(echo src/billing.ts)'],
['cat -n src/billing.ts','rm src/billing.ts'],
])('replacement output or mutation stays outside the closed display-tail form', (old,next)=>{
expect(reads(s=>{s.use.input.command=s.use.input.command.replace(old,next);})).toEqual(neither);
});
test('the native result must deliver exact ordered complete reads, even when Git hides a prefix failure',()=>{
for(const mutate of [
(s:ReturnType<typeof fresh>)=>{s.ack.is_error=true;},
(s:ReturnType<typeof fresh>)=>{s.ack.content='cat: src/billing.ts: No such file\n'+s.ack.content;},
(s:ReturnType<typeof fresh>)=>{s.ack.content=s.ack.content.replace('====\n','====\ncat: test/billing.test.ts: Permission denied\n');},
(s:ReturnType<typeof fresh>)=>{s.ack.content=s.row.files.source.content;},
(s:ReturnType<typeof fresh>)=>{s.ack.content=s.row.files.tests.content;},
(s:ReturnType<typeof fresh>)=>{s.ack.content=s.ack.content.replace("return { status: 'success', amount, currency };","return undefined;");},
(s:ReturnType<typeof fresh>)=>{const parts=s.ack.content.split('====');s.ack.content=parts[1]+'===='+parts[0]+'====';},
(s:ReturnType<typeof fresh>)=>{s.row.result.transcript[2].session_id='foreign';},
(s:ReturnType<typeof fresh>)=>{s.row.result.transcript[2].parent_tool_use_id='child';},
(s:ReturnType<typeof fresh>)=>{s.ack.tool_use_id='foreign';},
(s:ReturnType<typeof fresh>)=>{s.row.result.transcript.push(structuredClone(s.row.result.transcript[2]));},
])expect(reads(mutate)).toEqual(neither);
});
test('checkbox legends permit current synonyms, pair order, above/below placement and case',()=>{
for(const legend of ['Legend: [x] tested [ ] no test','Legend [x] covered by an existing test; [ ] no test reaches this path','Legend: [ ] untested | [X] covered'])expect(diagram(base.replace('Legend: [x] tested [ ] no test',legend))).toBe(true);
expect(diagram(base.replaceAll('[x]','[X]'))).toBe(true);
expect(diagram(base.replace('Legend: [x] tested [ ] no test\n','').replace('processPayment','Legend: [x] tested [ ] no test\nprocessPayment'))).toBe(true);
});
test.each(['','> Legend: [x] tested [ ] no test','"Legend: [x] tested [ ] no test"','Source: Legend: [x] tested [ ] no test','If approved, Legend: [x] tested [ ] no test','Legend: [x] untested [ ] covered','Legend: [x] tested [ ] covered','Legend: [x] tested [x] no test','Legend: [x] tested [ ] no test except refunds','Legend: [x] tested [ ] no test\nLegend: [x] untested [ ] covered'])('missing or contradictory checkbox key gives no diagram coverage: %s',legend=>{
expect(diagram(base.replace('Legend: [x] tested [ ] no test',legend))).toBe(false);
});
test('checkbox meanings cannot come from another block, stale key, or source declaration',()=>{
expect(diagram('```\nLegend: [x] tested [ ] no test\n```\n'+base.replace('Legend: [x] tested [ ] no test\n',''))).toBe(false);
for(const status of ['withdrawn','`no longer current`',"'superseded'",'“rejected”'])for(const boundary of ['\n','\nAssessment complete; '])expect(diagram(base.replace('\n```',boundary+'This legend is '+status+'.\n```'))).toBe(false);
for(const statement of [' This legend is withdrawn.','**This legend** is `no longer current`.','This legend applies only if approved.'])expect(diagram(base.replace('\n```','\n'+statement+'\n```'))).toBe(false);
expect(diagram(base.replace('\n```','\nEarlier reviewer said "This legend is withdrawn."\n```'))).toBe(true);
expect(diagram(base.replace('\n```','\n> Earlier note; This legend is withdrawn.\n```'))).toBe(true);
for(const prefix of ['Source:','Historical note:','Hypothetical:'])expect(diagram(base.replace('Legend:',prefix+'\nLegend:'))).toBe(false);
});
test('checkbox states retain final correction, function subtree and column ownership',()=>{
expect(diagram(base.replace('success [x]','success [ ] -> [x]'))).toBe(true);
expect(diagram(base.replace('refunded [ ]','refunded [x] → [ ]'))).toBe(true);
for(const [old,next]of [['success [x]','success [x] [ ]'],['refunded [ ]','refunded [ ] [x]'],['success [x]','success not [x]'],['refunded [ ]','refunded [ ] is incorrect'],['success [x]','success never covered [x]'],['refunded [ ]','refunded no coverage gaps [ ]'],['success [x]','success [x] -> [ ]'],['refunded [ ]','refunded [ ] → [x]'],['success [x]','success ├── [x]'],['refunded [ ]','refunded └── [ ]']])expect(diagram(base.replace(old!,next!))).toBe(false);
for(const name of ['processPayment','refundPayment'])expect(diagram(base.replace(name,'unrelated'))).toBe(false);
expect(diagram(base.replace('└── happy','otherFunction()\n└── happy'))).toBe(false);
expect(diagram(base.split('\n').map(l=>'> '+l).join('\n'))).toBe(false);
expect(diagram('````markdown\n'+base+'\n````')).toBe(false);
expect(diagram('Example:\n'+base)).toBe(false);
});
});
describe('coverage-diagram-legend-as', () => {
const captured = captured_coverage_diagram_legend_as;
const billing = fixture;
function verdict(output: string, index = 0) {
const row = captured.rows[index]!;
return coverageAuditVerdict({ ...row.result, output } as any, {
cwd: row.cwd,
source: { path: row.cwd + '/src/billing.ts', content: billing.files.source },
tests: { path: row.cwd + '/test/billing.test.ts', content: billing.files.tests },
});
}
const diagram = (output: string) => verdict(output).diagram;
const flat = (legend = 'Legend: [✔] tested [✘] GAP (no test)') => '```text\n' + legend + '\nprocessPayment(amount, currency)\n├──► return success [✔]\nrefundPayment(paymentId, reason)\n└──► return refunded [✘]\n```';
test('both exact public outputs contain the seeded diagram and retain actual native file delivery', () => {
for (let i = 0; i < captured.rows.length; i++) {
expect(verdict(captured.rows[i]!.result.output, i)).toEqual({ sourceRead: true, testsRead: true, diagram: true, passed: true, failures: [] });
}
expect(captured.provenance.originalAttemptOutcomes).toEqual(['failed', 'failed']);
expect(captured.provenance.paidOutcomesReclassified).toBe(false);
});
test('closed legend annotations preserve the same two meanings and arrow branch ownership', () => {
for (const legend of ['Legend: [✔] tested [✘] GAP', 'Legend: [✔] tested [✘] GAP (no test)', 'Legend: [✔] tested [✘] GAP ──► branch', 'Legend: [✔] tested [✘] GAP (no test) ──► branch']) {
expect(diagram(flat(legend))).toBe(true);
expect(diagram(flat(legend).replace(/^([├└]─+)►/gm, '$1'))).toBe(true);
expect(diagram(flat(legend).replaceAll('✔', '✓').replaceAll('✘', '✗'))).toBe(true);
}
});
test('extra legend explanations cannot invert, qualify or fabricate coverage meanings', () => {
for (const legend of ['', 'Legend: [✔] GAP [✘] tested', 'Legend: [✔] tested [✘] tested', 'Legend: [✔] tested [✘] GAP (not a gap)', 'Legend: [✔] tested [✘] GAP except refunds', 'Legend: [✔] tested [✘] GAP [✘] covered', 'Example: [✔] tested [✘] GAP', 'Legend: not [✔] tested [✘] GAP', 'Legend: [✔] tested [✘] GAP ──► covered']) {
expect(diagram(flat(legend))).toBe(false);
}
});
test('a branch status correction supplies its final state and ambiguous markers supply neither', () => {
expect(diagram(flat().replace('return refunded [✘]', 'return refunded [✔]→[✘]'))).toBe(true);
expect(diagram(flat().replace('return success [✔]', 'return success [✘]->[✔]'))).toBe(true);
expect(diagram(flat().replace('return success [✔]', 'return success [✔]→[✘]'))).toBe(false);
expect(diagram(flat().replace('return refunded [✘]', 'return refunded [✘]→[✔]'))).toBe(false);
expect(diagram(flat().replace('return success [✔]', 'return success [✔] [✘]'))).toBe(false);
expect(diagram(flat().replace('return refunded [✘]', 'return refunded [✘] [✔]'))).toBe(false);
});
test('literal labels cannot override a final or ambiguous bracketed symbol state', () => {
for (const [old, replacement] of [
['return success [✔]', 'return success TESTED [✔]→[✘]'],
['return refunded [✘]', 'return refunded UNTESTED [✘]→[✔]'],
['return success [✔]', 'return success TESTED [✔] [✘]'],
['return refunded [✘]', 'return refunded [GAP] [✘] [✔]'],
]) expect(diagram(flat().replace(old!, replacement!))).toBe(false);
});
test('each seeded function must own its own branch and legend in the same current diagram', () => {
for (const output of [
flat().replace('refundPayment', 'otherRefund'), flat().replace('processPayment', 'otherPayment'),
flat().replace('├──► return success [✔]', 'unrelatedHelper()\n├──► return success [✔]'),
flat().replace('└──► return refunded [✘]', 'unrelatedHelper()\n└──► return refunded [✘]'),
flat().split('\n').map(line => '> ' + line).join('\n'), '````markdown\n' + flat() + '\n````',
'Example:\n' + flat(), flat().replace('[✔] tested [✘] GAP (no test)', '[✔] tested (not covered) [✘] GAP'),
'```text\nLegend: [✔] tested [✘] GAP\n```\n' + flat(''),
]) expect(diagram(output)).toBe(false);
});
test('successful diagram parsing cannot replace successful capture or native file delivery', () => {
const row = captured.rows[0]!;
const files = { cwd: row.cwd, source: { path: row.cwd + '/src/billing.ts', content: billing.files.source }, tests: { path: row.cwd + '/test/billing.test.ts', content: billing.files.tests } };
for (const mutate of [
(r: any) => { r.exitReason = 'timeout'; }, (r: any) => { r.browseErrors = ['read failed']; },
(r: any) => { r.transcript = []; }, (r: any) => { r.transcript[2].message.content[0].is_error = true; },
(r: any) => { r.transcript[2].message.content[0].content = 'Both filenames were read'; },
]) {
const result = structuredClone(row.result); mutate(result);
const checked = coverageAuditVerdict(result as any, files);
expect(checked.diagram).toBe(true); expect(checked.passed).toBe(false);
}
});
});
describe('coverage-shell-display-aq', () => {
const path = posix;
const fixture = fixture_coverage_shell_display_aq;
const billing = billing_coverage_shell_display_aq;
function replay(row: typeof fixture.rows[number], command?: string) {
const transcript = structuredClone(row.transcript) as any[];
if (command !== undefined) transcript[1].message.content[0].input.command = command;
const cwd = transcript[0].cwd;
return coverageAuditReadEvidence(transcript, {
cwd, source: { path: path.join(cwd, 'src/billing.ts'), content: billing.files.source },
tests: { path: path.join(cwd, 'test/billing.test.ts'), content: billing.files.tests },
});
}
const command = (row: typeof fixture.rows[number]) => (row.transcript[1] as any).message.content[0].input.command as string;
describe('coverage reads with neighboring display commands', () => {
test('both exact failed AQ attempts delivered source and tests in their acknowledged Bash result', () => {
expect(fixture.provenance.actualPassedCases).toBe(0);
for (const row of fixture.rows) expect(replay(row)).toEqual({ sourceRead: true, testsRead: true });
});
test('a literal grep range and numeric Git log count do not own the delivered file bytes', () => {
const first = fixture.rows[0]!, second = fixture.rows[1]!;
expect(replay(first, command(first).replace('head -40', 'head -25'))).toEqual({ sourceRead: true, testsRead: true });
expect(replay(second, command(second).replace('log --oneline -3', 'log --oneline -12'))).toEqual({ sourceRead: true, testsRead: true });
});
test.each([
['awk action', (s: string) => s.replace("awk '/^### 3\\. Test review/,/^### 4\\./'", "awk 'BEGIN { system(\"cat fake\") }'")],
['awk output redirection', (s: string) => s.replace("awk '/^### 3\\. Test review/,/^### 4\\./'", "awk '/x/ { print > \"src/billing.ts\" }'")],
['shell substitution', (s: string) => s.replace('grep -n', 'grep -n "$(cat fake)"')],
['backtick execution', (s: string) => s.replace('grep -n', 'grep -n `cat fake`')],
['quoted injected command', (s: string) => s.replace('grep -n', 'grep -n "x"; printf fake; grep -n')],
['read hidden in a conditional', (s: string) => s.replace('cat -n src/billing.ts', 'false && cat -n src/billing.ts')],
['source-only filename', (s: string) => s.replace('cat -n src/billing.ts', "echo 'cat -n src/billing.ts'")],
] as const)('%s cannot borrow source read evidence', (_, mutate) => {
const row = fixture.rows[0]!;
expect(mutate(command(row))).not.toBe(command(row));
expect(replay(row, mutate(command(row))).sourceRead).toBe(false);
});
test.each([
'git log --output=src/billing.ts -3',
'git log --ext-diff -3',
'git log --format=%x00 -3',
'git log -3; printf fake',
])('unsupported Git command %s cannot borrow delivery', git => {
const row = fixture.rows[1]!;
expect(replay(row, command(row).replace('git log --oneline -3', git)).sourceRead).toBe(false);
});
test.each(['-f/tmp/other.awk', "'-f/tmp/other.awk'", "'--source=BEGIN {print \"fake\"}'"])(
'awk input %s cannot introduce another program', operand => {
const row = fixture.rows[0]!;
const changed = command(row).replace("Test review/,/^### 4\\./' plan-eng-review/sections/review-sections.md", "Test review/,/^### 4\\./' " + operand);
expect(changed).not.toBe(command(row));
expect(replay(row, changed).sourceRead).toBe(false);
});
test('successful command identity still requires the complete file and paired parent result', () => {
for (const row of fixture.rows) {
const missing = structuredClone(row) as any;
missing.transcript[2].message.content[0].content = 'src/billing.ts and test/billing.test.ts were read';
expect(replay(missing)).toEqual({ sourceRead: false, testsRead: false });
const failed = structuredClone(row) as any;
failed.transcript[2].message.content[0].is_error = true;
expect(replay(failed)).toEqual({ sourceRead: false, testsRead: false });
}
});
});
});