mirror of
https://github.com/garrytan/gstack.git
synced 2026-10-02 17:40:02 +02:00
touchfiles.test.ts now checks, per key, that the paid file's static test/helpers and test/fixtures closure (plus fixture paths it names in string literals) is covered, and names the file, path, chain and key to fix when it is not. Free *.test.ts files are no longer touchfiles, so editing a free replay test stops selecting paid evals: 950 entries removed, 653 real closure paths added. The hand-copied inventories go: periodic-fixture-selection, fake-impeccable-touchfiles and 45 per-file selection examples. Selection for the sample edits (plan-eng-review template, claude-pty-runner, plan-count-fixture, gstack-config) loses no case under either profile. CONTRIBUTING documents the rule and its lower bound.
986 lines
69 KiB
TypeScript
986 lines
69 KiB
TypeScript
import { describe, expect, test } from 'bun:test';
|
||
import * as path from 'node:path';
|
||
import fixture from './fixtures/coverage-audit-ae.json';
|
||
import ciDiagrams from './fixtures/coverage-audit-ci-diagrams.json';
|
||
import { coverageAuditVerdict } from './helpers/coverage-audit-evidence';
|
||
import { recordE2E } from './helpers/e2e-helpers';
|
||
import fixture_coverage_audit_af from './fixtures/coverage-audit-af.json';
|
||
import { posix } from 'node:path';
|
||
import { win32 } from 'node:path';
|
||
import { coverageAuditReadEvidence } from './helpers/coverage-audit-evidence';
|
||
import fixture_coverage_audit_aw from './fixtures/coverage-audit-aw.json';
|
||
import captured_coverage_audit_shell_legend_at from './fixtures/coverage-audit-shell-legend-at.json';
|
||
import fixture_coverage_checkbox_tail_av from './fixtures/coverage-checkbox-tail-av.json';
|
||
import captured_coverage_diagram_legend_as from './fixtures/coverage-diagram-legend-as.json';
|
||
import fixture_coverage_shell_display_aq from './fixtures/coverage-shell-display-aq.json';
|
||
import billing_coverage_shell_display_aq from './fixtures/coverage-audit-ae.json';
|
||
|
||
const clone = <T>(v:T):T => structuredClone(v);
|
||
const diagram = '```text\nsrc/billing.ts\n├── processPayment: happy path [TESTED]\n└── refundPayment [UNTESTED]\n```';
|
||
function synthetic() {
|
||
const cwd = '/tmp/coverage-audit-evidence-owned';
|
||
const files = {cwd, source:{path:path.join(cwd,'src/billing.ts'),content:fixture.files.source},
|
||
tests:{path:path.join(cwd,'test/billing.test.ts'),content:fixture.files.tests}};
|
||
const transcript:any[] = [{type:'system',subtype:'init',session_id:'parent',cwd}];
|
||
for (const [id,file] of Object.entries({source:files.source,tests:files.tests})) {
|
||
transcript.push({type:'assistant',session_id:'parent',parent_tool_use_id:null,message:{role:'assistant',content:[
|
||
{type:'tool_use',id,name:'Read',input:{file_path:file.path}},
|
||
]}});
|
||
transcript.push({type:'user',session_id:'parent',parent_tool_use_id:null,message:{role:'user',content:[
|
||
{type:'tool_result',tool_use_id:id,content:file.content},
|
||
]}});
|
||
}
|
||
return {files,result:{exitReason:'success',browseErrors:[],output:diagram,transcript} as any};
|
||
}
|
||
const verdict = (s:ReturnType<typeof synthetic>) => coverageAuditVerdict(s.result,s.files);
|
||
const block = (s:ReturnType<typeof synthetic>,i:number) => s.result.transcript[i].message.content[0];
|
||
|
||
describe('coverage audit native evidence',()=>{
|
||
test('literal cat operands preserve quoted whitespace for single and multiple owned paths', () => {
|
||
for (const quote of ["'", '"']) for (const flags of ['', '-n ', '-n -- ']) {
|
||
const s = synthetic();
|
||
s.files.cwd = '/repo with space';
|
||
s.files.source.path = s.files.cwd + '/src/billing source.ts';
|
||
s.files.tests.path = s.files.cwd + '/test/billing test.ts';
|
||
s.result.transcript[0].cwd = s.files.cwd;
|
||
for (const [i, file] of [[1, s.files.source], [3, s.files.tests]] as const) {
|
||
Object.assign(block(s, i), { name: 'Bash', input: { command: `cat ${flags}${quote}${file.path}${quote}` } });
|
||
}
|
||
expect(verdict(s)).toMatchObject({ sourceRead: true, testsRead: true });
|
||
block(s, 1).input.command = `cat ${flags}${quote}${s.files.source.path}${quote} ${quote}${s.files.tests.path}${quote}`;
|
||
block(s, 2).content = s.files.source.content + '\n' + s.files.tests.content;
|
||
s.result.transcript.splice(3);
|
||
expect(verdict(s)).toMatchObject({ sourceRead: true, testsRead: true });
|
||
for (const operands of [
|
||
`${quote}${s.files.source.path}${quote}suffix`,
|
||
`${quote}${s.files.source.path}${quote}${quote}${s.files.tests.path}${quote}`,
|
||
`${quote}${s.files.source.path}`, '"$SOURCE_FILE"', '`cat path`',
|
||
]) {
|
||
block(s, 1).input.command = `cat ${flags}${operands}`;
|
||
expect(verdict(s), operands).toMatchObject({ sourceRead: false, testsRead: false });
|
||
}
|
||
}
|
||
});
|
||
|
||
test('single word legend entries consume complete unqualified coverage and quality clauses', () => {
|
||
const s = synthetic();
|
||
const output = (legend: string) => '```text\nprocessPayment()\n└─ happy path [OK]\nrefundPayment()\n└─ happy path [GAP]\n' + legend + '\n```';
|
||
for (const legend of [
|
||
'Legend: [OK] covered\nLegend: [GAP] no test',
|
||
'Legend: ★★★ edges + errors ★★ happy path only ★ smoke [OK] tested\nLegend: [GAP] no test [→E2E] recommend integration test',
|
||
]) {
|
||
s.result.output = output(legend);
|
||
expect(verdict(s).diagram, legend).toBe(true);
|
||
for (const qualified of [
|
||
legend.replace('[OK] covered', '[OK] covered only if approved').replace('[OK] tested', '[OK] tested only if approved'),
|
||
legend.replace('[GAP] no test', '[GAP] no test except refunds'),
|
||
legend.replace('[GAP] no test', '[GAP] no test unless approved'),
|
||
legend.replace('[GAP] no test', 'hypothetical [GAP] no test'),
|
||
legend.replace('[OK]', 'not [OK]'),
|
||
legend + ' unknown qualifier',
|
||
]) {
|
||
s.result.output = output(qualified);
|
||
expect(verdict(s).diagram, qualified).toBe(false);
|
||
}
|
||
}
|
||
});
|
||
|
||
test('all four exact completed public attempts delivered both files and the seeded diagram',()=>{
|
||
expect(fixture.provenance.actualPassedCases).toBe(0);
|
||
for(const row of fixture.rows){
|
||
const files={cwd:row.cwd,source:{path:path.join(row.cwd,'src/billing.ts'),content:fixture.files.source},
|
||
tests:{path:path.join(row.cwd,'test/billing.test.ts'),content:fixture.files.tests}};
|
||
expect(coverageAuditVerdict(row.result as any,files)).toEqual({sourceRead:true,testsRead:true,diagram:true,passed:true,failures:[]});
|
||
}
|
||
});
|
||
test('both exact CI diagrams retain covered payment and missing refund paths', () => {
|
||
expect(ciDiagrams.provenance.recordedAttemptOutcomes).toEqual(['failed', 'failed']);
|
||
expect(ciDiagrams.provenance.paidOutcomesReclassified).toBe(false);
|
||
for (const row of ciDiagrams.diagrams) {
|
||
const s = synthetic(); s.result.output = row.text;
|
||
expect(verdict(s)).toEqual({ sourceRead: true, testsRead: true, diagram: true, passed: true, failures: [] });
|
||
}
|
||
});
|
||
const displayLegend = (legend: string, covered = '#', gap = ' ') => '```text\n' + legend + '\n'
|
||
+ 'processPayment(amount, currency)\n└─ valid return success [' + covered + ']\n'
|
||
+ 'refundPayment(paymentId, reason)\n└─ valid return refunded [' + gap + '] GAP\n```';
|
||
test('paid coverage diagrams accept a branch line without an arrowhead and a declared hash checkbox', () => {
|
||
for (const output of [
|
||
displayLegend('Legend: [✓] tested [✗] GAP (no test) ── branch', '✓', '✗'),
|
||
displayLegend('src/billing.ts — coverage map [#] tested [ ] GAP'),
|
||
displayLegend('Legend: [#] tested [ ] no test'),
|
||
displayLegend('src/billing.ts — coverage map [x] tested [ ] GAP', 'x'),
|
||
]) {
|
||
const s = synthetic(); s.result.output = output;
|
||
expect(verdict(s)).toEqual({ sourceRead: true, testsRead: true, diagram: true, passed: true, failures: [] });
|
||
}
|
||
});
|
||
// Exact public diagram from the completed, failed September 21 /review run.
|
||
// Its footer declares both symbol meanings without punctuation after Legend.
|
||
const symbolFooterDiagram = `\`\`\`
|
||
src/billing.ts test/billing.test.ts
|
||
======================================================================================
|
||
|
||
processPayment(amount, currency) describe('processPayment')
|
||
│
|
||
├── [✓] amount > 0 && currency in {USD, EUR} processes valid payment (L6-9)
|
||
│ → return { status: 'success', ... } (L5) processPayment(100, 'USD')
|
||
│
|
||
├── [✗] amount <= 0 ── NO TEST ──
|
||
│ → throw 'Invalid amount' (L3) gap: 0, negative values untested
|
||
│
|
||
└── [✗] currency not USD/EUR ── NO TEST ──
|
||
→ throw 'Unsupported currency' (L4) gap: 'GBP', '', lowercase 'usd'
|
||
|
||
|
||
refundPayment(paymentId, reason) (no describe block; not imported)
|
||
│
|
||
├── [✗] paymentId && reason truthy ── NO TEST ──
|
||
│ → return { status: 'refunded', ... } (L11) gap: happy path never exercised
|
||
│
|
||
├── [✗] !paymentId ── NO TEST ──
|
||
│ → throw 'Payment ID required' (L9) gap: '' / undefined untested
|
||
│
|
||
└── [✗] !reason ── NO TEST ──
|
||
→ throw 'Reason required' (L10) gap: '' / undefined untested
|
||
|
||
======================================================================================
|
||
Legend [✓] covered [✗] gap
|
||
|
||
Branches: 1 / 6 covered (17%)
|
||
Functions: 1 / 2 covered (50%)
|
||
Guard clauses tested: 0 / 4
|
||
\`\`\``;
|
||
test('the exact paid symbol footer may omit its colon', () => {
|
||
const s = synthetic(); s.result.output = symbolFooterDiagram;
|
||
expect(verdict(s)).toEqual({ sourceRead: true, testsRead: true, diagram: true, passed: true, failures: [] });
|
||
});
|
||
test('a colonless symbol footer still requires a current, affirmative, owned key', () => {
|
||
const key = 'Legend [✓] covered [✗] gap';
|
||
for (const replacement of ['', '> ' + key, '"' + key + '"', 'Example: ' + key,
|
||
'If approved: ' + key, key.replace('covered [✗] gap', 'gap [✗] covered'),
|
||
key.replace('[✗] gap', '[✗] covered'), key.replace('[✗]', '[✓]'),
|
||
key + ' except refunds', key + '\nLegend [✓] gap [✗] covered',
|
||
key + '\nThis legend is withdrawn.', key + '\nThis legend applies only if approved.',
|
||
]) {
|
||
const s = synthetic(); s.result.output = symbolFooterDiagram.replace(key, replacement);
|
||
expect(verdict(s).diagram, replacement).toBe(false);
|
||
}
|
||
for (const output of [
|
||
'\`\`\`text\n' + key + '\n\`\`\`\n' + symbolFooterDiagram.replace(key, ''),
|
||
symbolFooterDiagram.replaceAll('refundPayment', 'otherRefund'),
|
||
symbolFooterDiagram.replace('├── [✓] amount', '├── [✗] amount'),
|
||
'Example:\n' + symbolFooterDiagram, '\`\`\`\`markdown\n' + symbolFooterDiagram + '\n\`\`\`\`',
|
||
]) {
|
||
const s = synthetic(); s.result.output = output; expect(verdict(s).diagram).toBe(false);
|
||
}
|
||
});
|
||
test('hash checkbox and branch-line legends retain explicit local meanings and ownership', () => {
|
||
const caption = 'src/billing.ts — coverage map [#] tested [ ] GAP';
|
||
for (const legend of ['', '> ' + caption, '"' + caption + '"', 'Example: ' + caption,
|
||
'If approved: ' + caption, caption.replace('[#] tested [ ] GAP', '[#] GAP [ ] tested'),
|
||
caption.replace('[ ] GAP', '[ ] tested'), caption.replace('[ ] GAP', '[#] GAP'),
|
||
caption + ' except refunds', caption + '\nLegend: [#] untested [ ] covered',
|
||
caption + '\nLegend:[#] untested [ ] covered',
|
||
caption + '\nsrc/billing.ts — coverage map[#] untested [ ] covered',
|
||
caption + '\nThis legend is withdrawn.', caption + '\nThis legend applies only if approved.',
|
||
]) {
|
||
const s = synthetic(); s.result.output = displayLegend(legend); expect(verdict(s).diagram).toBe(false);
|
||
}
|
||
const valid = displayLegend(caption);
|
||
for (const output of [
|
||
'```text\n' + caption + '\n```\n' + displayLegend(''),
|
||
valid.replace('processPayment', 'otherPayment'), valid.replace('refundPayment', 'otherRefund'),
|
||
valid.replace('return success [#]', 'return success not [#]'),
|
||
valid.replace('return success [#]', 'return success [#] -> [ ]'),
|
||
valid.replace('return refunded [ ]', 'return refunded [ ] -> [#]'),
|
||
valid.replace('return refunded [ ]', 'return refunded [ ] [#]'),
|
||
valid.replace('return success [#]', 'return success ├─ [#]'),
|
||
'````markdown\n' + valid + '\n````', 'Example:\n' + valid,
|
||
displayLegend('Legend: [✓] tested [✗] GAP ── covered', '✓', '✗'),
|
||
displayLegend('Legend: [✓] tested [✗] GAP ── branch except refunds', '✓', '✗'),
|
||
]) {
|
||
const s = synthetic(); s.result.output = output; expect(verdict(s).diagram).toBe(false);
|
||
}
|
||
const s = synthetic(); s.result.output = valid; s.result.transcript = [];
|
||
expect(verdict(s).diagram).toBe(true); expect(verdict(s).passed).toBe(false);
|
||
});
|
||
test('CI symbol legends remain current, unambiguous and owned by their diagram', () => {
|
||
for (const row of ciDiagrams.diagrams) {
|
||
const text = row.text, key = text.split('\n').find(line => line.startsWith('Legend:'))!;
|
||
for (const replacement of ['', '> ' + key, 'Source: ' + key, key + ' except refunds',
|
||
key.replace(/covered(?: by a test)?/, 'untested'),
|
||
key + '\nLegend: [✓] GAP [✗] covered', key + '\n [✓] GAP [✗] covered',
|
||
...['Sample:', 'Example legend:', 'Illustration:'].map(label => label + '\n' + key)]) {
|
||
const s = synthetic(); s.result.output = text.replace(key, replacement);
|
||
expect(verdict(s).diagram, replacement).toBe(false);
|
||
}
|
||
for (const status of ['This legend is withdrawn.', 'Assessment complete; This legend is `no longer current`.',
|
||
'**This legend** is “rejected”.', 'This legend applies only if approved.']) {
|
||
const s = synthetic(); s.result.output = text.replace(/\n```$/, '\n' + status + '\n```');
|
||
expect(verdict(s).diagram, status).toBe(false);
|
||
}
|
||
for (const output of ['Example:\n' + text, '````markdown\n' + text + '\n````',
|
||
text.replace(/^```[^\n]*/, '```json'), '```\n' + key + '\n```\n' + text.replace(key, '')]) {
|
||
const s = synthetic(); s.result.output = output; expect(verdict(s).diagram, output).toBe(false);
|
||
}
|
||
const s = synthetic(); s.result.output = text.replace(/\n```$/, '\nEarlier reviewer said "This legend is withdrawn."\n```');
|
||
expect(verdict(s).diagram).toBe(true);
|
||
}
|
||
});
|
||
test('six-column annotations cannot borrow sibling, prose or parallel-column markers', () => {
|
||
const text = '```\nLegend: [✓] covered by a test [✗] GAP — no test exercises this path\n'
|
||
+ 'processPayment(amount, currency)\n└── happy return success\n [✓] covered\n'
|
||
+ 'refundPayment(paymentId, reason)\n└── return refunded\n [✗] GAP\n```';
|
||
for (const output of [text.replace(' [✓]', 'unrelatedPayment()\n [✓]'),
|
||
text.replace(' [✓]', ' Earlier example:\n [✓]'),
|
||
text.replace(' [✓]', ' [✓]'),
|
||
text.replace(' [✓]', ' [✗]'), text.replace(' [✗]', ' [✓]'),
|
||
text.replace('└── happy return success\n [✓]', '└── happy return success ├── [✓]')]) {
|
||
const s = synthetic(); s.result.output = output; expect(verdict(s).diagram).toBe(false);
|
||
}
|
||
const s = synthetic(); s.result.output = text; expect(verdict(s).passed).toBe(true);
|
||
});
|
||
test.each([
|
||
'```text\nsrc/billing.ts\n├── refundPayment [UNTESTED]\n└── processPayment: happy path [TESTED]\n```',
|
||
'src/billing.ts\n├── processPayment: happy path [TESTED]\n└── refundPayment [UNTESTED]',
|
||
])('function order and optional fencing do not change valid coverage evidence: %s', output=>{
|
||
const s=synthetic();s.result.output=output;expect(verdict(s).diagram).toBe(true);
|
||
});
|
||
test('direct Read, literal cat/sed and delivered native line gutters are valid',()=>{
|
||
for(const command of ['cat -n src/billing.ts',"sed -n '1,200p' 'src/billing.ts'",'cat -- "src/billing.ts"']){
|
||
const s=synthetic();Object.assign(block(s,1),{name:'Bash',input:{command}});
|
||
block(s,2).content=fixture.files.source.split('\n').map((line,i)=>`${i+1}\t${line}`).join('\n');
|
||
expect(verdict(s).passed).toBe(true);
|
||
}
|
||
const s=synthetic();block(s,2).content=[{type:'text',text:fixture.files.source.split('\n').map((line,i)=>`${i+1}→${line}`).join('\n')}];
|
||
expect(verdict(s).passed).toBe(true);
|
||
});
|
||
function mixedDisplay(context: boolean) {
|
||
const s = synthetic();
|
||
const command = context
|
||
? 'cat review/specialists/testing.md && echo ==== SRC ==== && cat -n src/billing.ts && echo ==== TEST ==== && cat -n test/billing.test.ts && echo ==== GIT ==== && git log --oneline main..HEAD; git diff main --stat'
|
||
: 'cat -n test/billing.test.ts && git log --oneline main..feature/billing 2>/dev/null; git diff main...feature/billing --stat 2>/dev/null';
|
||
const numbered = (body: string) => body.replace(/\n$/, '').split('\n').map((line, index) => `${index + 1}\t${line}`).join('\n');
|
||
if (context) s.result.transcript.splice(1, 2);
|
||
const use = s.result.transcript.at(-2).message.content[0];
|
||
const result = s.result.transcript.at(-1).message.content[0];
|
||
Object.assign(use, {name: 'Bash', input: {command}});
|
||
result.content = context
|
||
? '# Testing Specialist Review Checklist\n\nCoverage Gaps\n==== SRC ====\n' + numbered(s.files.source.content)
|
||
+ '\n==== TEST ====\n' + numbered(s.files.tests.content) + '\n==== GIT ===='
|
||
: numbered(s.files.tests.content);
|
||
return {s, use, result};
|
||
}
|
||
test('mixed Git display tails retain separately delivered files and numbered reads after context', () => {
|
||
// Shell forms from the two failed 2026-09-20 paid /review captures.
|
||
for (const context of [false, true]) expect(verdict(mixedDisplay(context).s).passed).toBe(true);
|
||
});
|
||
test('mixed display reads retain ordered bodies and successful parent ownership', () => {
|
||
for (const context of [false, true]) for (const mutate of [
|
||
(x: ReturnType<typeof mixedDisplay>) => { x.result.is_error = true; },
|
||
(x: ReturnType<typeof mixedDisplay>) => { x.result.content = 'test/billing.test.ts was read'; },
|
||
(x: ReturnType<typeof mixedDisplay>) => { x.result.content = x.result.content.replace(/.*import \{ describe.*\n/, ''); },
|
||
(x: ReturnType<typeof mixedDisplay>) => { x.s.result.transcript.at(-1).session_id = 'foreign'; },
|
||
(x: ReturnType<typeof mixedDisplay>) => { x.s.result.transcript.at(-1).parent_tool_use_id = 'child'; },
|
||
(x: ReturnType<typeof mixedDisplay>) => { x.result.tool_use_id = 'unpaired'; },
|
||
(x: ReturnType<typeof mixedDisplay>) => { x.s.result.transcript.push(clone(x.s.result.transcript.at(-1))); },
|
||
]) {
|
||
const x = mixedDisplay(context); mutate(x); expect(verdict(x.s).testsRead).toBe(false);
|
||
}
|
||
for (const context of [false, true]) for (const suffix of [
|
||
'git diff main --output=src/billing.ts --stat', 'git diff main --ext-diff --stat',
|
||
'git diff main --stat > output.txt', 'git diff main --stat || echo ok',
|
||
]) {
|
||
const x = mixedDisplay(context); x.use.input.command = x.use.input.command.replace(/git diff[^;]+$/, suffix);
|
||
expect(verdict(x.s).testsRead).toBe(false);
|
||
}
|
||
for (const prefix of ['cat ../foreign.md', 'cat --help.md', 'cat /foreign.md', 'cat "$CONTEXT"', 'cat review/specialists/testing.md | head -2',
|
||
'false', 'python3 -c "pass"', 'echo -e "replacement"', 'cat review/specialists/testing.md; false']) {
|
||
const x = mixedDisplay(true); x.use.input.command = x.use.input.command.replace('cat review/specialists/testing.md', prefix);
|
||
expect(verdict(x.s).sourceRead).toBe(false); expect(verdict(x.s).testsRead).toBe(false);
|
||
}
|
||
const x = mixedDisplay(true); x.result.content = x.result.content.replace('==== SRC ====', '==== OTHER ====');
|
||
expect(verdict(x.s).sourceRead).toBe(false); expect(verdict(x.s).testsRead).toBe(false);
|
||
const repeated = mixedDisplay(true); repeated.result.content += '\n==== SRC ====';
|
||
expect(verdict(repeated.s).sourceRead).toBe(false); expect(verdict(repeated.s).testsRead).toBe(false);
|
||
const missing = mixedDisplay(true); missing.result.content = missing.result.content.slice(missing.result.content.indexOf('==== SRC ===='));
|
||
expect(verdict(missing.s).sourceRead).toBe(false); expect(verdict(missing.s).testsRead).toBe(false);
|
||
});
|
||
function boundDisplay(kind: 'and-log' | 'quoted-grep') {
|
||
const s = synthetic();
|
||
const numbered = (body: string) => body.replace(/\n$/, '').split('\n').map((line, index) => `${index + 1}\t${line}`).join('\n');
|
||
// Exact commands from the two completed, failed 2026-09-20 bound reruns.
|
||
const command = kind === 'and-log'
|
||
? 'cat -n src/billing.ts && echo ==== && cat -n test/billing.test.ts && echo ==== && git log --oneline main..HEAD && git diff main --stat'
|
||
: "grep -n -i 'diagram\\|coverage\\|tested\\|gap' review/SKILL.md | head -60; echo ======SRC; cat -n src/billing.ts; echo ======TEST; cat -n test/billing.test.ts; echo ======DIFF; git diff main...HEAD --stat";
|
||
s.result.transcript.splice(3, 2);
|
||
const use = block(s, 1), result = block(s, 2);
|
||
Object.assign(use, {name: 'Bash', input: {command}});
|
||
result.content = kind === 'and-log'
|
||
? numbered(s.files.source.content) + '\n====\n' + numbered(s.files.tests.content) + '\n===='
|
||
: '119: Test coverage gaps for stated requirements\n======SRC\n' + numbered(s.files.source.content)
|
||
+ '\n======TEST\n' + numbered(s.files.tests.content) + '\n======DIFF';
|
||
return {s, use, result};
|
||
}
|
||
test.each(['and-log', 'quoted-grep'] as const)('complete parent reads survive closed neighboring displays: %s', kind => {
|
||
expect(verdict(boundDisplay(kind).s)).toEqual({sourceRead:true, testsRead:true, diagram:true, passed:true, failures:[]});
|
||
});
|
||
test('neighboring log and quoted grep displays cannot replace complete owned delivery', () => {
|
||
for (const kind of ['and-log', 'quoted-grep'] as const) for (const mutate of [
|
||
(x: ReturnType<typeof boundDisplay>) => { x.result.is_error = true; },
|
||
(x: ReturnType<typeof boundDisplay>) => { x.result.content = 'src/billing.ts and test/billing.test.ts were read'; },
|
||
(x: ReturnType<typeof boundDisplay>) => { x.s.result.transcript[2].session_id = 'foreign'; },
|
||
(x: ReturnType<typeof boundDisplay>) => { x.s.result.transcript[2].parent_tool_use_id = 'child'; },
|
||
(x: ReturnType<typeof boundDisplay>) => { x.s.result.transcript[1].parent_tool_use_id = 'child'; },
|
||
(x: ReturnType<typeof boundDisplay>) => { x.result.tool_use_id = 'unpaired'; },
|
||
(x: ReturnType<typeof boundDisplay>) => { x.s.result.transcript.push(clone(x.s.result.transcript[2])); },
|
||
]) {
|
||
const x = boundDisplay(kind); mutate(x);
|
||
expect(verdict(x.s).sourceRead).toBe(false); expect(verdict(x.s).testsRead).toBe(false);
|
||
}
|
||
for (const kind of ['and-log', 'quoted-grep'] as const) for (const [key, line] of [
|
||
['sourceRead', /.*export function processPayment.*\n/], ['testsRead', /.*import \{ describe.*\n/],
|
||
] as const) {
|
||
const x = boundDisplay(kind); x.result.content = x.result.content.replace(line, '');
|
||
expect(verdict(x.s)[key]).toBe(false); expect(verdict(x.s).passed).toBe(false);
|
||
}
|
||
});
|
||
test('closed neighboring log and grep grammars reject unsafe lookalikes', () => {
|
||
for (const display of [
|
||
'git log --oneline main..HEAD --output=src/billing.ts', 'git log --oneline main..HEAD --format=%B',
|
||
'git log --oneline main..HEAD --ext-diff', 'git log --oneline main..HEAD > output.txt',
|
||
'git log --oneline "main..HEAD"', 'git log --oneline main..HEAD || echo ok',
|
||
]) {
|
||
const x = boundDisplay('and-log'); x.use.input.command = x.use.input.command.replace('git log --oneline main..HEAD', display);
|
||
expect(verdict(x.s).sourceRead).toBe(false); expect(verdict(x.s).testsRead).toBe(false);
|
||
}
|
||
for (const display of [
|
||
'grep -n -i "$(touch sentinel)" review/SKILL.md | head -60',
|
||
'grep -n -i "`touch sentinel`" review/SKILL.md | head -60',
|
||
"grep -n -i 'diagram\\|coverage' --help | head -60",
|
||
"grep -n -i 'diagram\\|coverage' review/SKILL.md > output.txt",
|
||
"grep -n -i 'diagram\\|coverage' review/SKILL.md | python3 -c 'pass'",
|
||
"grep -n -i 'diagram\\ncoverage' review/SKILL.md | head -60",
|
||
]) {
|
||
const x = boundDisplay('quoted-grep'); x.use.input.command = x.use.input.command.replace(/^[^;]+/, display);
|
||
expect(verdict(x.s).sourceRead).toBe(false); expect(verdict(x.s).testsRead).toBe(false);
|
||
}
|
||
});
|
||
test('each exact source and test file must be successfully delivered',()=>{
|
||
for(const mutate of [
|
||
(s:any)=>{block(s,2).content='src/billing.ts was read';},
|
||
(s:any)=>{block(s,2).content=fixture.files.source.split('\n').slice(0,3).join('\n');},
|
||
(s:any)=>{block(s,2).is_error=true;},
|
||
(s:any)=>{block(s,3).input.file_path=s.files.source.path;},
|
||
(s:any)=>{block(s,4).content=fixture.files.source;},
|
||
(s:any)=>{block(s,1).input.file_path=path.join(s.files.cwd,'other/billing.ts');},
|
||
(s:any)=>{s.files.tests.path=s.files.source.path;},
|
||
]){const s=synthetic();mutate(s);expect(verdict(s).passed).toBe(false);}
|
||
});
|
||
test('unpaired, repeated, child and foreign events cannot supply parent file evidence',()=>{
|
||
for(const mutate of [
|
||
(s:any)=>{s.result.transcript.splice(1,1);},
|
||
(s:any)=>{[s.result.transcript[1],s.result.transcript[2]]=[s.result.transcript[2],s.result.transcript[1]];},
|
||
(s:any)=>{s.result.transcript[2].session_id='foreign';},
|
||
(s:any)=>{s.result.transcript[2].parent_tool_use_id='agent';},
|
||
(s:any)=>{s.result.transcript[1].parent_tool_use_id='agent';},
|
||
(s:any)=>{s.result.transcript[2].message.role='assistant';},
|
||
(s:any)=>{s.result.transcript.push(clone(s.result.transcript[2]));},
|
||
(s:any)=>{s.result.transcript.push(clone(s.result.transcript[1]));},
|
||
(s:any)=>{s.result.transcript[0].cwd+='/sibling';},
|
||
(s:any)=>{s.result.transcript.push(clone(s.result.transcript[0]));},
|
||
(s:any)=>{s.result.transcript[0].session_id='foreign';},
|
||
(s:any)=>{s.result.transcript[0].type='user';},
|
||
(s:any)=>{s.result.transcript.shift();},
|
||
]){const s=synthetic();mutate(s);expect(verdict(s).passed).toBe(false);}
|
||
});
|
||
test('quoted metadata, counters and undeclared shell reads do not substitute for actual delivery',()=>{
|
||
for(const command of [
|
||
"echo 'cat src/billing.ts'",'false && cat src/billing.ts','cat src/billing.ts | head -2',
|
||
'cd ../sibling; cat src/billing.ts',"if true; then cat src/billing.ts; fi",'cat "$SOURCE"',
|
||
"cat <<'EOF'\ncat src/billing.ts\nEOF",'f() {\ncat src/billing.ts\n}',
|
||
'(\ncat src/billing.ts\n)',
|
||
]){const s=synthetic();Object.assign(block(s,1),{name:'Bash',input:{command}});expect(verdict(s).passed).toBe(false);}
|
||
const s=synthetic();s.result.toolCalls=[{tool:'Read',input:{file_path:s.files.source.path}},{tool:'Read',input:{file_path:s.files.tests.path}}];
|
||
s.result.transcript=[s.result.transcript[0],{type:'assistant',session_id:'parent',message:{role:'assistant',content:[{type:'text',text:JSON.stringify(s.result.transcript.slice(1))}]}}];
|
||
expect(verdict(s).sourceRead).toBe(false);expect(verdict(s).testsRead).toBe(false);
|
||
});
|
||
test('a commented read cannot borrow printed bytes; quoted hash paths remain literal',()=>{
|
||
const s=synthetic();
|
||
const command=`printf '${Buffer.from(fixture.files.source).toString('base64')}' | base64 -d; # only printed bytes; cat src/billing.ts`;
|
||
Object.assign(block(s,1),{name:'Bash',input:{command}});
|
||
expect(verdict(s).sourceRead).toBe(false);
|
||
const quoted=synthetic();quoted.files.source.path=path.join(quoted.files.cwd,'src/billing#branch.ts');
|
||
Object.assign(block(quoted,1),{name:'Bash',input:{command:"cat 'src/billing#branch.ts'"}});
|
||
expect(verdict(quoted).sourceRead).toBe(true);
|
||
});
|
||
test('completion and tool errors remain final gate failures despite genuine delivery',()=>{
|
||
for(const exitReason of ['timeout','exit_code_1']){const s=synthetic();s.result.exitReason=exitReason;expect(verdict(s).passed).toBe(false);}
|
||
const s=synthetic();s.result.browseErrors=['read failed'];expect(verdict(s).passed).toBe(false);
|
||
});
|
||
test('coverage markers must belong to the seeded payment and refund functions',()=>{
|
||
for(const output of [
|
||
diagram.replace('[TESTED]','[UNTESTED]'),diagram.replace('[UNTESTED]','[TESTED]'),
|
||
diagram.replace('processPayment','processPaymentExample'),diagram.replace('refundPayment','refundPaymentExample'),
|
||
'```\n├── processPayment: happy path [TESTED]\n└── refundPayment [TESTED]\n└── unrelatedPayment [UNTESTED]\n```',
|
||
'```\n├── processPayment: happy path [TESTED]\n```\n```\n└── refundPayment [UNTESTED]\n```',
|
||
]){const s=synthetic();s.result.output=output;expect(verdict(s).diagram).toBe(false);}
|
||
});
|
||
test('quoted and nested source examples are not the generated coverage diagram',()=>{
|
||
for(const output of [diagram.split('\n').map(line=>'> '+line).join('\n'), '````markdown\n'+diagram+'\n````']){
|
||
const s=synthetic();s.result.output=output;expect(verdict(s).diagram).toBe(false);
|
||
}
|
||
});
|
||
test.each([
|
||
diagram.replace('[UNTESTED]','[NOT UNTESTED]'),
|
||
diagram.replace('[UNTESTED]','[UNTESTED] is false; this function is fully covered.'),
|
||
'Example only; this diagram is not the audit result.\n'+diagram,
|
||
])('negated gaps and explicitly labeled examples are not audit findings: %s', output=>{
|
||
const s=synthetic();s.result.output=output;expect(verdict(s).diagram).toBe(false);
|
||
});
|
||
test('collector receives exactly the asserted verdict even when process exit succeeded',()=>{
|
||
for(const valid of [true,false]){
|
||
const s=synthetic();if(!valid)s.result.output='No diagram produced.';
|
||
Object.assign(s.result,{toolCalls:[],duration:1,costEstimate:{estimatedCost:0,turnsUsed:1,estimatedTokens:1}});
|
||
const v=verdict(s),entries:any[]=[];
|
||
recordE2E({addTest:(entry:any)=>entries.push(entry)} as any,'coverage','fixture',s.result,{passed:v.passed,error:v.failures.length?v.failures.join('; '):undefined});
|
||
expect(entries).toHaveLength(1);expect(entries[0].passed).toBe(valid);expect(entries[0].error).toBe(valid?undefined:v.failures.join('; '));
|
||
}
|
||
});
|
||
});
|
||
|
||
describe('coverage-audit-af', () => {
|
||
const fixture = fixture_coverage_audit_af;
|
||
const actual = (index: number) => structuredClone(fixture.rows[index]!);
|
||
const files = (row: typeof fixture.rows[number]) => ({cwd:row.cwd,
|
||
source:{path:`${row.cwd}/src/billing.ts`,content:fixture.files.source},
|
||
tests:{path:`${row.cwd}/test/billing.test.ts`,content:fixture.files.tests}});
|
||
for (let i=0;i<fixture.rows.length;i++) test(`AF exact public coverage audit ${i+1} binds delivered files and seeded diagram`, () => {
|
||
const row=actual(i); expect(coverageAuditVerdict(row.result, files(row))).toEqual({sourceRead:true,testsRead:true,diagram:true,passed:true,failures:[]});
|
||
});
|
||
|
||
function delivered(command: string, mutate?: (events: any[]) => void) {
|
||
const row=actual(2), session=row.sessionId;
|
||
const transcript:any[]=[
|
||
{type:'system',subtype:'init',session_id:session,cwd:row.cwd},
|
||
{type:'assistant',session_id:session,parent_tool_use_id:null,message:{role:'assistant',content:[{type:'tool_use',id:'read-pair',name:'Bash',input:{command}}]}},
|
||
{type:'user',session_id:session,parent_tool_use_id:null,message:{role:'user',content:[{type:'tool_result',tool_use_id:'read-pair',is_error:false,content:fixture.files.source+'\n----\n'+fixture.files.tests}]}},
|
||
];
|
||
mutate?.(transcript);
|
||
return coverageAuditVerdict({...row.result,transcript},files(row));
|
||
}
|
||
const both = 'cat -n src/billing.ts && cat -n test/billing.test.ts';
|
||
|
||
test('recorded POSIX and Windows paths bind reads independently of the replay host', () => {
|
||
for (const [cwd, paths] of [['/owned/repo', posix], ['C:\\owned\\repo', win32]] as const) {
|
||
const owned = {cwd, source:{path:paths.join(cwd,'src/billing.ts'),content:fixture.files.source},
|
||
tests:{path:paths.join(cwd,'test/billing.test.ts'),content:fixture.files.tests}};
|
||
const transcript = [
|
||
{type:'system',subtype:'init',session_id:'owned',cwd},
|
||
{type:'assistant',session_id:'owned',message:{role:'assistant',content:[{type:'tool_use',id:'pair',name:'Bash',input:{command:both}}]}},
|
||
{type:'user',session_id:'owned',message:{role:'user',content:[{type:'tool_result',tool_use_id:'pair',is_error:false,content:fixture.files.source+'\n'+fixture.files.tests}]}},
|
||
];
|
||
expect(coverageAuditReadEvidence(transcript,owned)).toEqual({sourceRead:true,testsRead:true});
|
||
expect(coverageAuditReadEvidence(transcript,{...owned,source:{...owned.source,path:paths.join(cwd,'../foreign.ts')}}))
|
||
.toEqual({sourceRead:false,testsRead:false});
|
||
expect(coverageAuditReadEvidence(transcript,{...owned,source:{...owned.source,path:cwd+paths.sep+'src'+paths.sep+'..'+paths.sep+'src'+paths.sep+'billing.ts'}}))
|
||
.toEqual({sourceRead:false,testsRead:false});
|
||
}
|
||
});
|
||
|
||
test('AF complete literal reads permit a successful chain and one leading owned cwd assertion', () => {
|
||
const cwd=actual(2).cwd;
|
||
for (const command of [both, `cd ${cwd}; cat -n src/billing.ts; cat -n test/billing.test.ts`, `cd '${cwd}' && ${both}`, 'cat -n src/billing.ts; echo ----; cat -n test/billing.test.ts']) {
|
||
const v=delivered(command); expect(v.sourceRead).toBe(true); expect(v.testsRead).toBe(true); expect(v.passed).toBe(true);
|
||
}
|
||
});
|
||
|
||
test('AF the new conditional-chain grammar conservatively rejects mixed separators', () => {
|
||
const v=delivered(`cd ${actual(2).cwd}; ${both}`);
|
||
expect(v.sourceRead).toBe(false); expect(v.testsRead).toBe(false);
|
||
});
|
||
|
||
test('AF read recognition rejects foreign or midstream cwd changes and nonliteral targets', () => {
|
||
const cwd=actual(2).cwd;
|
||
for (const command of [`cd /foreign; ${both}`, `cat -n src/billing.ts; cd /foreign; cat -n test/billing.test.ts`,
|
||
`cat -n src/billing.ts; cd ${cwd}; cat -n test/billing.test.ts`, `cd "$PWD"; ${both}`, `cd ${cwd}/..; ${both}`]) {
|
||
const v=delivered(command); expect(v.sourceRead).toBe(false); expect(v.testsRead).toBe(false);
|
||
}
|
||
});
|
||
|
||
test('AF a printed, conditional or skipped read cannot borrow delivered-looking file contents', () => {
|
||
for (const command of [`false && ${both}`, `if false; then ${both}; fi`, `echo '${both}'`, `exit; ${both}`,
|
||
`# ${both}`, `cat <<'EOF'\n${both}\nEOF`, `(${both})`, `f() { ${both}; }`, `printf '%s' '${both}'`,
|
||
`printf expected; false && ${both}; true`, `${both} > result.txt`]) {
|
||
const v=delivered(command); expect(v.sourceRead).toBe(false); expect(v.testsRead).toBe(false);
|
||
}
|
||
});
|
||
|
||
test('AF an owned cwd does not authorize mutations or interpreters around a read', () => {
|
||
for (const neighbor of ['rm -f src/billing.ts', 'python3 -c "pass"', 'echo fake > src/billing.ts',
|
||
'grep data backup.txt | tee src/billing.ts', 'git diff --output=src/billing.ts',
|
||
"git diff '--output=src/billing.ts'", "git diff --output'='src/billing.ts",
|
||
'git diff --out=src/billing.ts', 'git diff --ext-diff']) {
|
||
const result = delivered(`cd ${actual(2).cwd}; ${neighbor}; cat src/billing.ts; cat test/billing.test.ts`);
|
||
expect(result.sourceRead).toBe(false); expect(result.testsRead).toBe(false);
|
||
}
|
||
});
|
||
|
||
test('AF added command forms retain exact parent request/result success and delivered-content binding', () => {
|
||
const mutations:Array<(events:any[])=>void>=[
|
||
e=>{e[0].cwd='/foreign';}, e=>{e[2].session_id='foreign';},
|
||
e=>{e[1].parent_tool_use_id='child';}, e=>{e[2].message.content[0].is_error=true;},
|
||
e=>{e[2].message.content[0].tool_use_id='unpaired';},
|
||
e=>{e[2].message.content[0].content='The two filenames were read.';},
|
||
];
|
||
for (const mutate of mutations) { const v=delivered(both,mutate); expect(v.sourceRead).toBe(false); expect(v.testsRead).toBe(false); }
|
||
const onlySource=delivered(both,e=>{e[2].message.content[0].content=fixture.files.source;});
|
||
expect(onlySource.sourceRead).toBe(true); expect(onlySource.testsRead).toBe(false);
|
||
});
|
||
|
||
const flat = (legend = 'Legend: [✓] tested [✗] GAP') => `\`\`\`text\n${legend}\nprocessPayment(amount, currency)\n├── [✓] happy path USD\nrefundPayment(paymentId, reason)\n└── [✗] happy path refund\n\`\`\``;
|
||
function diagram(output:string) { const row=actual(0);return coverageAuditVerdict({...row.result,output},files(row)).diagram; }
|
||
|
||
test('AF flat function roots and same-block legend symbols preserve seeded coverage ownership', () => {
|
||
expect(diagram(flat())).toBe(true);
|
||
expect(diagram(flat().replaceAll('✓','✔').replaceAll('✗','✘'))).toBe(true);
|
||
expect(diagram(flat().replace('processPayment(amount, currency)\n├── [✓] happy path USD\nrefundPayment(paymentId, reason)\n└── [✗] happy path refund',
|
||
'refundPayment(paymentId, reason)\n├── [✗] happy path refund\nprocessPayment(amount, currency)\n└── [✓] happy path USD'))).toBe(true);
|
||
});
|
||
|
||
test('AF symbol-only markers need an unambiguous legend in their own diagram block', () => {
|
||
for (const output of [flat(''),flat('Legend: [✓] GAP [✗] tested'),flat('Legend: [✓] tested [✗] tested'),
|
||
flat('Legend: [✓] tested [✗] GAP [✗] tested'),
|
||
`\`\`\`text\nLegend: [✓] tested [✗] GAP\n\`\`\`\n${flat('')}`]) expect(diagram(output)).toBe(false);
|
||
});
|
||
|
||
test('AF flat roots cannot borrow another function subtree or a quoted/example diagram', () => {
|
||
for (const output of [
|
||
flat().replace('├── [✓] happy path USD','├── untested amount guard\nunrelatedHelper()\n└── [✓] happy path USD'),
|
||
flat().replace('refundPayment(paymentId, reason)','unrelatedRefund(paymentId, reason)'),
|
||
flat().replace('processPayment(amount, currency)','processPaymentOther(amount, currency)'),
|
||
flat().split('\n').map(line=>'> '+line).join('\n'),
|
||
'````markdown\n'+flat()+'\n````',
|
||
flat().replace('Legend:','Example diagram:\nLegend:'),
|
||
]) expect(diagram(output)).toBe(false);
|
||
});
|
||
|
||
test('AF a legend cannot override an explicitly negated marker on its own branch', () => {
|
||
for (const output of [
|
||
flat().replace('[✗] happy path refund','not [✗] happy path refund'),
|
||
flat().replace('[✗] happy path refund','[✗] is false; this branch is tested'),
|
||
flat().replace('[✓] happy path USD','not [✓] happy path USD'),
|
||
]) expect(diagram(output)).toBe(false);
|
||
});
|
||
test('AF symbol gaps retain affirmative legend ownership and reject same-branch contradictions', () => {
|
||
for (const output of [
|
||
flat().replace('[✗] happy path refund', '[✗] happy path refund (marker is incorrect; this branch is fully tested)'),
|
||
flat().replace('[✗] happy path refund', '[✗] happy path refund — no coverage gap exists'),
|
||
flat('An unproven hypothesis: [✓] tested [✗] GAP'),
|
||
flat("The source says '[✓] tested [✗] GAP'"),
|
||
]) expect(diagram(output)).toBe(false);
|
||
expect(diagram(flat())).toBe(true);
|
||
expect(diagram(flat('[✓] tested [✗] GAP'))).toBe(true);
|
||
expect(diagram(flat('src/billing.ts — test coverage map [✓] tested [✗] GAP'))).toBe(true);
|
||
});
|
||
});
|
||
|
||
describe('coverage-audit-aw', () => {
|
||
const fixture = fixture_coverage_audit_aw;
|
||
const fresh=(n=0)=>{
|
||
const r=structuredClone(fixture.reads[n]!),files={cwd:r.cwd,source:{path:r.cwd+'/src/billing.ts',content:fixture.source},tests:{path:r.cwd+'/test/billing.test.ts',content:fixture.tests}};
|
||
const transcript:any[]=[{type:'system',subtype:'init',session_id:r.sessionId,cwd:r.cwd},{type:'assistant',session_id:r.sessionId,parent_tool_use_id:null,message:{role:'assistant',content:[{type:'tool_use',id:r.toolUseId,name:'Bash',input:{command:r.command}}]}},{type:'user',session_id:r.sessionId,parent_tool_use_id:null,message:{role:'user',content:[{type:'tool_result',tool_use_id:r.toolUseId,is_error:false,content:r.outputExcerpt}]}}];
|
||
return {files,transcript};
|
||
};
|
||
const reads=(x:ReturnType<typeof fresh>)=>coverageAuditReadEvidence(x.transcript,x.files);
|
||
const diagram=(text:string)=>{const x=fresh();return coverageAuditVerdict({exitReason:'success',browseErrors:[],output:text,transcript:x.transcript} as any,x.files).diagram;};
|
||
describe('Coverage audit owned display composition and marker continuations',()=>{
|
||
test.each([0,1,2,3])('credits exact complete public file delivery %i',n=>{
|
||
expect(fixture.provenance.actualCollectorFailuresRetained).toBe(true);expect(reads(fresh(n))).toEqual({sourceRead:true,testsRead:true});
|
||
});
|
||
test.each(['failed result','foreign session','foreign cwd','foreign tool id','sidechain','missing result','repeated result','partial body','forged body'])('rejects %s',form=>{
|
||
for(let n=0;n<4;n++){const x=fresh(n),e=x.transcript[2],b=e.message.content[0];if(form==='failed result')b.is_error=true;else if(form==='foreign session')e.session_id='foreign';else if(form==='foreign cwd')x.transcript[0].cwd+='/other';else if(form==='foreign tool id')b.tool_use_id='foreign';else if(form==='sidechain')e.parent_tool_use_id='parent';else if(form==='missing result')x.transcript.pop();else if(form==='repeated result')x.transcript.push(structuredClone(e));else if(form==='partial body')b.content=b.content.replace(/.*(?:export function processPayment|import \{ describe).*\n/g,'');else b.content='src/billing.ts and test/billing.test.ts were read';expect(reads(x)).toEqual({sourceRead:false,testsRead:false});}
|
||
});
|
||
test.each(['foreign paths','printf forgery','echo escape forgery','expansion','double quoted expansion','awk execution','changed ordered prefix'])('rejects unsupported or unowned command: %s',form=>{
|
||
const n=form==='awk execution'?2:form==='changed ordered prefix'?3:0,x=fresh(n),u=x.transcript[1].message.content[0];u.input.command=form==='foreign paths'?u.input.command.replaceAll('src/billing.ts','other/billing.ts').replaceAll('test/billing.test.ts','other/billing.test.ts'):form==='printf forgery'?"printf 'fixture body'":form==='echo escape forgery'?"echo -e 'fake\\nbody'":form==='expansion'?u.input.command+'; echo $(cat source)':form==='double quoted expansion'?u.input.command+'; echo "$HOME"':form==='awk execution'?u.input.command.replace('{f=1}','{system("cat forged") }'):u.input.command.replace('=== src/billing.ts ===','=== other.ts ===');expect(reads(x)).toEqual({sourceRead:false,testsRead:false});
|
||
});
|
||
test.each([0,1])('accepts the exact public current diagram %i',n=>expect(diagram(fixture.diagrams[n]!.text)).toBe(true));
|
||
test.each(['missing key','inverted checkbox key','withdrawn key','foreign function','quoted source','not covered','not missing'])('rejects contradictory or unowned checkbox coverage: %s',form=>{
|
||
const text=fixture.diagrams[0]!.text;const changed=form==='missing key'?text.replace(/^Legend:.*\n/m,''):form==='inverted checkbox key'?text.replace('[x] tested [ ] GAP','[x] untested [ ] tested'):form==='withdrawn key'?text.replace('src/billing.ts\n│','This legend is withdrawn.\nsrc/billing.ts\n│'):form==='foreign function'?text.replaceAll('refundPayment','otherPayment'):form==='quoted source'?'Example only:\n'+text:form==='not covered'?text.replace("[x] 'processes valid payment'","[ ] GAP"):text.replaceAll('[ ] GAP','[x] tested');expect(diagram(changed)).toBe(false);
|
||
});
|
||
test('continuations keep their own row and cannot borrow from prose or a distant column',()=>{
|
||
const text=fixture.diagrams[1]!.text;
|
||
expect(diagram(text.replace('│ [✓] billing.test.ts:6', '│ Earlier example:\n│ [✓] billing.test.ts:6'))).toBe(false);
|
||
expect(diagram(text.replace('│ [✓] billing.test.ts:6', ' [✓] billing.test.ts:6'))).toBe(false);
|
||
expect(diagram(text.replace('│ [✓] billing.test.ts:6', '│ [✗] billing.test.ts:6'))).toBe(false);
|
||
expect(diagram(text.replaceAll('[✗] GAP','[✓] tested').replace('[✗] untested (GAP)','[✗] untested (GAP)'))).toBe(false);
|
||
});
|
||
});
|
||
});
|
||
|
||
describe('coverage-audit-shell-legend-at', () => {
|
||
const captured = captured_coverage_audit_shell_legend_at;
|
||
const both = { sourceRead: true, testsRead: true };
|
||
const neither = { sourceRead: false, testsRead: false };
|
||
function owned(index: number) {
|
||
const row = structuredClone(captured[index]!) as any;
|
||
const useEvent = row.result.transcript.find((event: any) => event.message?.content.some((block: any) =>
|
||
block.type === 'tool_use' && block.name === 'Bash' && block.input.command.includes('cat -n src/billing.ts')));
|
||
const use = useEvent.message.content.find((block: any) => block.type === 'tool_use' && block.name === 'Bash' && block.input.command.includes('cat -n src/billing.ts'));
|
||
const resultEvent = row.result.transcript.find((event: any) => event.message?.content.some((block: any) => block.type === 'tool_result' && block.tool_use_id === use.id));
|
||
row.result.transcript = [row.result.transcript.find((event: any) => event.type === 'system' && event.subtype === 'init'), useEvent, resultEvent];
|
||
return { row, use, resultEvent, delivered: resultEvent.message.content.find((block: any) => block.tool_use_id === use.id) };
|
||
}
|
||
function reads(index: number, mutate?: (s: ReturnType<typeof owned>) => void) {
|
||
const s = owned(index); mutate?.(s);
|
||
return coverageAuditReadEvidence(s.row.result.transcript, s.row.files);
|
||
}
|
||
const flat = (legend = 'Legend [ OK ] covered [ GAP ] no test') =>
|
||
'```text\nprocessPayment(amount, currency)\n├── happy path return success [ OK ]\nrefundPayment(paymentId, reason)\n└── happy path return refunded [ GAP ]\n' + legend + '\n```';
|
||
const diagram = (output: string) => coverageAuditVerdict({ ...captured[1]!.result, output } as any, captured[1]!.files).diagram;
|
||
|
||
test('exact public AT first, retry and engineering outputs retain all required native evidence', () => {
|
||
expect(captured.map(row => row.recordedPassed)).toEqual([false, false, true]);
|
||
for (const row of captured) expect(coverageAuditVerdict(row.result as any, row.files)).toEqual({ ...both, diagram: true, passed: true, failures: [] });
|
||
expect(reads(0)).toEqual(both); expect(reads(1)).toEqual(both);
|
||
});
|
||
|
||
test('literal grep display options and filename captions do not own source bytes', () => {
|
||
for (const flags of ['-n', '-n -i', '-n -B1 -A200', '-n -i -B3 -A40']) {
|
||
expect(reads(0, s => { s.use.input.command = s.use.input.command.replace('-n -i -B3 -A40', flags); })).toEqual(both);
|
||
}
|
||
for (const replace of ['echo \'=== another-file.md ===\'', 'echo "--- src/billing.ts ---"', 'echo']) {
|
||
expect(reads(1, s => { s.use.input.command = s.use.input.command.replace('echo "=== testing.md ==="', replace); })).toEqual(both);
|
||
}
|
||
expect(reads(1, s => { s.use.input.command = s.use.input.command.replace('git diff main --stat', 'git diff HEAD~1 --stat'); })).toEqual(both);
|
||
});
|
||
|
||
test('escaped grep patterns keep a closed flag and literal operand grammar', () => {
|
||
for (const replacement of ['-n -i -B3 -A40 -f other', '-n -i --include=*', '-n -B-1', '-n -A100000', '-n -i -B3 -A40; false']) {
|
||
expect(reads(0, s => { s.use.input.command = s.use.input.command.replace('-n -i -B3 -A40', replacement); })).toEqual(neither);
|
||
}
|
||
for (const operand of ['-f/tmp/foreign', '"-f/tmp/foreign"', 'review/SKILL.md --include=*']) {
|
||
expect(reads(0, s => { s.use.input.command = s.use.input.command.replace('review/SKILL.md |', operand + ' |'); })).toEqual(neither);
|
||
}
|
||
});
|
||
|
||
test('successful conditional display paths reject execution, substitutions and hidden failure', () => {
|
||
for (const replacement of [
|
||
'echo -e "=== testing.md ==="', 'printf "=== testing.md ==="', 'echo "$(cat fake)"', 'echo `cat fake`',
|
||
'echo "=== testing.md ==="; false', 'false || echo "=== testing.md ==="', 'unknown',
|
||
'echo "cat -n src/billing.ts"', 'echo "=== testing.md ===\\nreplacement"',
|
||
'cd ../sibling', 'env PATH=/tmp cat fake', 'echo "=== testing.md ===" > src/billing.ts',
|
||
]) expect(reads(1, s => { s.use.input.command = s.use.input.command.replace('echo "=== testing.md ==="', replacement); })).toEqual(neither);
|
||
for (const command of ['git diff --ext-diff --stat', 'git diff main --output=src/billing.ts --stat', 'git -c core.pager=evil diff main --stat', 'git diff --no-index main --stat', 'git diff main --stat || echo ok']) {
|
||
expect(reads(1, s => { s.use.input.command = s.use.input.command.replace('git diff main --stat', command); })).toEqual(neither);
|
||
}
|
||
});
|
||
|
||
test('a valid display path still requires one complete successful owned delivery', () => {
|
||
for (const index of [0, 1]) for (const mutate of [
|
||
(s: ReturnType<typeof owned>) => { s.delivered.is_error = true; },
|
||
(s: ReturnType<typeof owned>) => { s.delivered.content = 'src/billing.ts and test/billing.test.ts were read'; },
|
||
(s: ReturnType<typeof owned>) => { s.delivered.content = s.row.files.source.content.slice(0, 80); },
|
||
(s: ReturnType<typeof owned>) => { s.resultEvent.session_id = 'foreign'; },
|
||
(s: ReturnType<typeof owned>) => { s.resultEvent.parent_tool_use_id = 'child'; },
|
||
(s: ReturnType<typeof owned>) => { s.delivered.tool_use_id = 'foreign'; },
|
||
(s: ReturnType<typeof owned>) => { s.row.result.transcript.push(structuredClone(s.resultEvent)); },
|
||
(s: ReturnType<typeof owned>) => { s.use.input.command = s.use.input.command.replace('cat -n src/billing.ts', 'echo src/billing.ts').replace('cat -n test/billing.test.ts', 'echo test/billing.test.ts'); },
|
||
]) expect(reads(index, mutate)).toEqual(neither);
|
||
expect(reads(1, s => { s.delivered.content = s.row.files.source.content; })).toEqual({ sourceRead: true, testsRead: false });
|
||
expect(reads(1, s => { s.delivered.content = s.row.files.tests.content; })).toEqual({ sourceRead: false, testsRead: true });
|
||
});
|
||
|
||
test('text statuses use the declared local meanings with whitespace and either pair order', () => {
|
||
for (const legend of ['Legend [ OK ] covered [ GAP ] no test', 'Legend: [OK] tested | [GAP] untested', 'Legend: [ GAP ] no test; [ OK ] covered']) expect(diagram(flat(legend))).toBe(true);
|
||
expect(diagram(flat().replaceAll('[ OK ]', '[OK]').replaceAll('[ GAP ]', '[GAP]'))).toBe(true);
|
||
expect(diagram(flat().replace('Legend [ OK ] covered [ GAP ] no test\n', '').replace('processPayment', 'Legend [ OK ] covered [ GAP ] no test\nprocessPayment'))).toBe(true);
|
||
});
|
||
|
||
test('missing, malformed, contradictory or foreign text legends cannot grant coverage', () => {
|
||
for (const legend of ['', '> Legend [ OK ] covered [ GAP ] no test', '"Legend [ OK ] covered [ GAP ] no test"',
|
||
'Example: Legend [ OK ] covered [ GAP ] no test', 'If enabled, Legend [ OK ] covered [ GAP ] no test',
|
||
'Legend [ OK ] no test [ GAP ] covered', 'Legend [ OK ] covered [ GAP ] covered',
|
||
'Legend [ OK ] covered [ OK ] no test', 'Legend [ OK ] covered [ GAP ] no test except refunds',
|
||
'Legend [ OK ] covered [ GAP ] no test\nLegend [ OK ] no test [ GAP ] covered',
|
||
]) expect(diagram(flat(legend))).toBe(false);
|
||
expect(diagram('```text\nLegend [ OK ] covered [ GAP ] no test\n```\n' + flat(''))).toBe(false);
|
||
for (const status of ['cancelled', 'canceled', 'rejected', 'retracted', 'withdrawn', "'withdrawn'", '‘superseded’', '`no longer current`', '"not current"']) {
|
||
expect(diagram(flat('Legend [ OK ] covered [ GAP ] no test\nThis legend is ' + status + '.'))).toBe(false);
|
||
}
|
||
expect(diagram(flat('Legend [ OK ] covered [ GAP ] no test\n> An old note said: "This legend is withdrawn."'))).toBe(true);
|
||
});
|
||
|
||
test('text marker corrections grant only the final unambiguous owned row state', () => {
|
||
expect(diagram(flat().replace('success [ OK ]', 'success [ GAP ] -> [ OK ]'))).toBe(true);
|
||
expect(diagram(flat().replace('refunded [ GAP ]', 'refunded [ OK ] → [ GAP ]'))).toBe(true);
|
||
for (const [old, replacement] of [
|
||
['success [ OK ]', 'success COVERED [ OK ] → [ GAP ]'],
|
||
['refunded [ GAP ]', 'refunded UNTESTED [ GAP ] -> [ OK ]'],
|
||
['success [ OK ]', 'success COVERED [ OK ] [ GAP ]'],
|
||
['refunded [ GAP ]', 'refunded [GAP] [ GAP ] [ OK ]'],
|
||
['success [ OK ]', 'success not [ OK ]'], ['refunded [ GAP ]', 'refunded [ GAP ] is incorrect'],
|
||
['success [ OK ]', 'success not covered [ OK ]'], ['refunded [ GAP ]', 'refunded no coverage gaps [ GAP ]'],
|
||
]) expect(diagram(flat().replace(old!, replacement!))).toBe(false);
|
||
});
|
||
|
||
test('text coverage markers retain function, subtree, column and source ownership', () => {
|
||
for (const output of [
|
||
flat().replace('processPayment', 'otherPayment'), flat().replace('refundPayment', 'otherRefund'),
|
||
flat().replace('├── happy', 'otherFunction()\n├── happy'), flat().replace('└── happy', 'otherFunction()\n└── happy'),
|
||
flat().replace('success [ OK ]', 'success ├── [ OK ]'), flat().replace('refunded [ GAP ]', 'refunded └── [ GAP ]'),
|
||
flat().split('\n').map(line => '> ' + line).join('\n'), '````markdown\n' + flat() + '\n````', 'Example:\n' + flat(),
|
||
]) expect(diagram(output)).toBe(false);
|
||
});
|
||
});
|
||
|
||
describe('coverage-checkbox-tail-av', () => {
|
||
const fixture = fixture_coverage_checkbox_tail_av;
|
||
const both={sourceRead:true,testsRead:true}, neither={sourceRead:false,testsRead:false};
|
||
const fresh=(i=0)=>{const row=structuredClone(fixture.attempts[i]!) as any;return{row,use:row.result.transcript[1].message.content[0],ack:row.result.transcript[2].message.content[0]};};
|
||
const reads=(mutate:(s:ReturnType<typeof fresh>)=>void=()=>{})=>{const s=fresh();mutate(s);return coverageAuditReadEvidence(s.row.result.transcript,s.row.files);};
|
||
const base='```text\nprocessPayment(amount, currency)\n├── happy return success [x]\nrefundPayment(paymentId, reason)\n└── happy return refunded [ ]\nLegend: [x] tested [ ] no test\n```';
|
||
const diagram=(output:string)=>{const {row}=fresh();return coverageAuditVerdict({...row.result,output},row.files).diagram;};
|
||
|
||
test('both exact public attempts now provide their delivered files and owned checkbox diagram',()=>{
|
||
expect(fixture.provenance.paidOutcomesReclassified).toBe(false);
|
||
for(const row of fixture.attempts){expect(row.provenance.recordedPassed).toBe(false);expect(coverageAuditVerdict(row.result as any,row.files)).toEqual({...both,diagram:true,passed:true,failures:[]});}
|
||
});
|
||
|
||
test('mixed display tail accepts only the two ordered owned reads and literal separators',()=>{
|
||
expect(reads()).toEqual(both);
|
||
for(const revision of ['HEAD','HEAD~1','main..HEAD'])expect(reads(s=>{s.use.input.command=s.use.input.command.replace('main..HEAD',revision).replace('diff main','diff '+revision);})).toEqual(both);
|
||
expect(reads(s=>{s.use.input.command=s.use.input.command.replaceAll('echo ====','echo ----');s.ack.content=s.ack.content.replaceAll('====','----');})).toEqual(both);
|
||
expect(reads(s=>{s.use.input.command=s.use.input.command.replace('src/billing.ts',"'src/billing.ts'").replace('test/billing.test.ts','"test/billing.test.ts"');})).toEqual(both);
|
||
expect(reads(s=>{s.ack.content=[{type:'text',text:s.ack.content}];})).toEqual(both);
|
||
expect(reads(s=>{s.ack.content+='\nabc1234 harmless commit subject\n src/billing.ts | 2 ++\n 1 file changed, 2 insertions(+)';})).toEqual(both);
|
||
});
|
||
|
||
test.each([
|
||
'false && cat -n src/billing.ts && cat -n test/billing.test.ts && git log --oneline main..HEAD; git diff main --stat',
|
||
'cat -n src/billing.ts && false && cat -n test/billing.test.ts && git log --oneline main..HEAD; git diff main --stat',
|
||
'cat -n fake.ts && cat -n test/billing.test.ts && git log --oneline main..HEAD; git diff main --stat',
|
||
'cat -n src/billing.ts && cat -n src/billing.ts && git log --oneline main..HEAD; git diff main --stat',
|
||
'cat -n src/billing.ts && cat -n test/billing.test.ts; git log --oneline main..HEAD; git diff main --stat',
|
||
'cat -n src/billing.ts; cat -n test/billing.test.ts && git log --oneline main..HEAD; git diff main --stat',
|
||
])('a failed, skipped, duplicate or unrelated read cannot borrow display-tail success: %s',command=>{
|
||
expect(reads(s=>{s.use.input.command=command;})).toEqual(neither);
|
||
});
|
||
|
||
test.each([
|
||
['git log --oneline main..HEAD','git log --format=%B main..HEAD'],
|
||
['git log --oneline main..HEAD','git log --oneline --output=src/billing.ts main..HEAD'],
|
||
['git log --oneline main..HEAD','git -c core.pager=evil log --oneline main..HEAD'],
|
||
['git diff main --stat','git diff --ext-diff main --stat'],
|
||
['git diff main --stat','git diff --no-index main --stat'],
|
||
['git diff main --stat','git diff main --stat; echo extra'],
|
||
['git diff main --stat','git diff main --stat > src/billing.ts'],
|
||
['echo ====','echo replacement'],['echo ====','printf ===='],['echo ====','echo -e "\\nreplacement"'],
|
||
['cat -n src/billing.ts','cat -n src/billing.ts > test/billing.test.ts'],
|
||
['cat -n src/billing.ts','cat -n $(echo src/billing.ts)'],
|
||
['cat -n src/billing.ts','rm src/billing.ts'],
|
||
])('replacement output or mutation stays outside the closed display-tail form', (old,next)=>{
|
||
expect(reads(s=>{s.use.input.command=s.use.input.command.replace(old,next);})).toEqual(neither);
|
||
});
|
||
|
||
test('the native result must deliver exact ordered complete reads, even when Git hides a prefix failure',()=>{
|
||
for(const mutate of [
|
||
(s:ReturnType<typeof fresh>)=>{s.ack.is_error=true;},
|
||
(s:ReturnType<typeof fresh>)=>{s.ack.content='cat: src/billing.ts: No such file\n'+s.ack.content;},
|
||
(s:ReturnType<typeof fresh>)=>{s.ack.content=s.ack.content.replace('====\n','====\ncat: test/billing.test.ts: Permission denied\n');},
|
||
(s:ReturnType<typeof fresh>)=>{s.ack.content=s.row.files.source.content;},
|
||
(s:ReturnType<typeof fresh>)=>{s.ack.content=s.row.files.tests.content;},
|
||
(s:ReturnType<typeof fresh>)=>{s.ack.content=s.ack.content.replace("return { status: 'success', amount, currency };","return undefined;");},
|
||
(s:ReturnType<typeof fresh>)=>{const parts=s.ack.content.split('====');s.ack.content=parts[1]+'===='+parts[0]+'====';},
|
||
(s:ReturnType<typeof fresh>)=>{s.row.result.transcript[2].session_id='foreign';},
|
||
(s:ReturnType<typeof fresh>)=>{s.row.result.transcript[2].parent_tool_use_id='child';},
|
||
(s:ReturnType<typeof fresh>)=>{s.ack.tool_use_id='foreign';},
|
||
(s:ReturnType<typeof fresh>)=>{s.row.result.transcript.push(structuredClone(s.row.result.transcript[2]));},
|
||
])expect(reads(mutate)).toEqual(neither);
|
||
});
|
||
|
||
test('checkbox legends permit current synonyms, pair order, above/below placement and case',()=>{
|
||
for(const legend of ['Legend: [x] tested [ ] no test','Legend [x] covered by an existing test; [ ] no test reaches this path','Legend: [ ] untested | [X] covered'])expect(diagram(base.replace('Legend: [x] tested [ ] no test',legend))).toBe(true);
|
||
expect(diagram(base.replaceAll('[x]','[X]'))).toBe(true);
|
||
expect(diagram(base.replace('Legend: [x] tested [ ] no test\n','').replace('processPayment','Legend: [x] tested [ ] no test\nprocessPayment'))).toBe(true);
|
||
});
|
||
|
||
test.each(['','> Legend: [x] tested [ ] no test','"Legend: [x] tested [ ] no test"','Source: Legend: [x] tested [ ] no test','If approved, Legend: [x] tested [ ] no test','Legend: [x] untested [ ] covered','Legend: [x] tested [ ] covered','Legend: [x] tested [x] no test','Legend: [x] tested [ ] no test except refunds','Legend: [x] tested [ ] no test\nLegend: [x] untested [ ] covered'])('missing or contradictory checkbox key gives no diagram coverage: %s',legend=>{
|
||
expect(diagram(base.replace('Legend: [x] tested [ ] no test',legend))).toBe(false);
|
||
});
|
||
|
||
test('checkbox meanings cannot come from another block, stale key, or source declaration',()=>{
|
||
expect(diagram('```\nLegend: [x] tested [ ] no test\n```\n'+base.replace('Legend: [x] tested [ ] no test\n',''))).toBe(false);
|
||
for(const status of ['withdrawn','`no longer current`',"'superseded'",'“rejected”'])for(const boundary of ['\n','\nAssessment complete; '])expect(diagram(base.replace('\n```',boundary+'This legend is '+status+'.\n```'))).toBe(false);
|
||
for(const statement of [' This legend is withdrawn.','**This legend** is `no longer current`.','This legend applies only if approved.'])expect(diagram(base.replace('\n```','\n'+statement+'\n```'))).toBe(false);
|
||
expect(diagram(base.replace('\n```','\nEarlier reviewer said "This legend is withdrawn."\n```'))).toBe(true);
|
||
expect(diagram(base.replace('\n```','\n> Earlier note; This legend is withdrawn.\n```'))).toBe(true);
|
||
for(const prefix of ['Source:','Historical note:','Hypothetical:'])expect(diagram(base.replace('Legend:',prefix+'\nLegend:'))).toBe(false);
|
||
});
|
||
|
||
test('checkbox states retain final correction, function subtree and column ownership',()=>{
|
||
expect(diagram(base.replace('success [x]','success [ ] -> [x]'))).toBe(true);
|
||
expect(diagram(base.replace('refunded [ ]','refunded [x] → [ ]'))).toBe(true);
|
||
for(const [old,next]of [['success [x]','success [x] [ ]'],['refunded [ ]','refunded [ ] [x]'],['success [x]','success not [x]'],['refunded [ ]','refunded [ ] is incorrect'],['success [x]','success never covered [x]'],['refunded [ ]','refunded no coverage gaps [ ]'],['success [x]','success [x] -> [ ]'],['refunded [ ]','refunded [ ] → [x]'],['success [x]','success ├── [x]'],['refunded [ ]','refunded └── [ ]']])expect(diagram(base.replace(old!,next!))).toBe(false);
|
||
for(const name of ['processPayment','refundPayment'])expect(diagram(base.replace(name,'unrelated'))).toBe(false);
|
||
expect(diagram(base.replace('└── happy','otherFunction()\n└── happy'))).toBe(false);
|
||
expect(diagram(base.split('\n').map(l=>'> '+l).join('\n'))).toBe(false);
|
||
expect(diagram('````markdown\n'+base+'\n````')).toBe(false);
|
||
expect(diagram('Example:\n'+base)).toBe(false);
|
||
});
|
||
});
|
||
|
||
describe('coverage-diagram-legend-as', () => {
|
||
const captured = captured_coverage_diagram_legend_as;
|
||
const billing = fixture;
|
||
|
||
function verdict(output: string, index = 0) {
|
||
const row = captured.rows[index]!;
|
||
return coverageAuditVerdict({ ...row.result, output } as any, {
|
||
cwd: row.cwd,
|
||
source: { path: row.cwd + '/src/billing.ts', content: billing.files.source },
|
||
tests: { path: row.cwd + '/test/billing.test.ts', content: billing.files.tests },
|
||
});
|
||
}
|
||
const diagram = (output: string) => verdict(output).diagram;
|
||
const flat = (legend = 'Legend: [✔] tested [✘] GAP (no test)') => '```text\n' + legend + '\nprocessPayment(amount, currency)\n├──► return success [✔]\nrefundPayment(paymentId, reason)\n└──► return refunded [✘]\n```';
|
||
|
||
test('both exact public outputs contain the seeded diagram and retain actual native file delivery', () => {
|
||
for (let i = 0; i < captured.rows.length; i++) {
|
||
expect(verdict(captured.rows[i]!.result.output, i)).toEqual({ sourceRead: true, testsRead: true, diagram: true, passed: true, failures: [] });
|
||
}
|
||
expect(captured.provenance.originalAttemptOutcomes).toEqual(['failed', 'failed']);
|
||
expect(captured.provenance.paidOutcomesReclassified).toBe(false);
|
||
});
|
||
|
||
test('closed legend annotations preserve the same two meanings and arrow branch ownership', () => {
|
||
for (const legend of ['Legend: [✔] tested [✘] GAP', 'Legend: [✔] tested [✘] GAP (no test)', 'Legend: [✔] tested [✘] GAP ──► branch', 'Legend: [✔] tested [✘] GAP (no test) ──► branch']) {
|
||
expect(diagram(flat(legend))).toBe(true);
|
||
expect(diagram(flat(legend).replace(/^([├└]─+)►/gm, '$1'))).toBe(true);
|
||
expect(diagram(flat(legend).replaceAll('✔', '✓').replaceAll('✘', '✗'))).toBe(true);
|
||
}
|
||
});
|
||
|
||
test('extra legend explanations cannot invert, qualify or fabricate coverage meanings', () => {
|
||
for (const legend of ['', 'Legend: [✔] GAP [✘] tested', 'Legend: [✔] tested [✘] tested', 'Legend: [✔] tested [✘] GAP (not a gap)', 'Legend: [✔] tested [✘] GAP except refunds', 'Legend: [✔] tested [✘] GAP [✘] covered', 'Example: [✔] tested [✘] GAP', 'Legend: not [✔] tested [✘] GAP', 'Legend: [✔] tested [✘] GAP ──► covered']) {
|
||
expect(diagram(flat(legend))).toBe(false);
|
||
}
|
||
});
|
||
|
||
test('a branch status correction supplies its final state and ambiguous markers supply neither', () => {
|
||
expect(diagram(flat().replace('return refunded [✘]', 'return refunded [✔]→[✘]'))).toBe(true);
|
||
expect(diagram(flat().replace('return success [✔]', 'return success [✘]->[✔]'))).toBe(true);
|
||
expect(diagram(flat().replace('return success [✔]', 'return success [✔]→[✘]'))).toBe(false);
|
||
expect(diagram(flat().replace('return refunded [✘]', 'return refunded [✘]→[✔]'))).toBe(false);
|
||
expect(diagram(flat().replace('return success [✔]', 'return success [✔] [✘]'))).toBe(false);
|
||
expect(diagram(flat().replace('return refunded [✘]', 'return refunded [✘] [✔]'))).toBe(false);
|
||
});
|
||
|
||
test('literal labels cannot override a final or ambiguous bracketed symbol state', () => {
|
||
for (const [old, replacement] of [
|
||
['return success [✔]', 'return success TESTED [✔]→[✘]'],
|
||
['return refunded [✘]', 'return refunded UNTESTED [✘]→[✔]'],
|
||
['return success [✔]', 'return success TESTED [✔] [✘]'],
|
||
['return refunded [✘]', 'return refunded [GAP] [✘] [✔]'],
|
||
]) expect(diagram(flat().replace(old!, replacement!))).toBe(false);
|
||
});
|
||
|
||
test('each seeded function must own its own branch and legend in the same current diagram', () => {
|
||
for (const output of [
|
||
flat().replace('refundPayment', 'otherRefund'), flat().replace('processPayment', 'otherPayment'),
|
||
flat().replace('├──► return success [✔]', 'unrelatedHelper()\n├──► return success [✔]'),
|
||
flat().replace('└──► return refunded [✘]', 'unrelatedHelper()\n└──► return refunded [✘]'),
|
||
flat().split('\n').map(line => '> ' + line).join('\n'), '````markdown\n' + flat() + '\n````',
|
||
'Example:\n' + flat(), flat().replace('[✔] tested [✘] GAP (no test)', '[✔] tested (not covered) [✘] GAP'),
|
||
'```text\nLegend: [✔] tested [✘] GAP\n```\n' + flat(''),
|
||
]) expect(diagram(output)).toBe(false);
|
||
});
|
||
|
||
test('successful diagram parsing cannot replace successful capture or native file delivery', () => {
|
||
const row = captured.rows[0]!;
|
||
const files = { cwd: row.cwd, source: { path: row.cwd + '/src/billing.ts', content: billing.files.source }, tests: { path: row.cwd + '/test/billing.test.ts', content: billing.files.tests } };
|
||
for (const mutate of [
|
||
(r: any) => { r.exitReason = 'timeout'; }, (r: any) => { r.browseErrors = ['read failed']; },
|
||
(r: any) => { r.transcript = []; }, (r: any) => { r.transcript[2].message.content[0].is_error = true; },
|
||
(r: any) => { r.transcript[2].message.content[0].content = 'Both filenames were read'; },
|
||
]) {
|
||
const result = structuredClone(row.result); mutate(result);
|
||
const checked = coverageAuditVerdict(result as any, files);
|
||
expect(checked.diagram).toBe(true); expect(checked.passed).toBe(false);
|
||
}
|
||
});
|
||
});
|
||
|
||
describe('coverage-shell-display-aq', () => {
|
||
const path = posix;
|
||
const fixture = fixture_coverage_shell_display_aq;
|
||
const billing = billing_coverage_shell_display_aq;
|
||
function replay(row: typeof fixture.rows[number], command?: string) {
|
||
const transcript = structuredClone(row.transcript) as any[];
|
||
if (command !== undefined) transcript[1].message.content[0].input.command = command;
|
||
const cwd = transcript[0].cwd;
|
||
return coverageAuditReadEvidence(transcript, {
|
||
cwd, source: { path: path.join(cwd, 'src/billing.ts'), content: billing.files.source },
|
||
tests: { path: path.join(cwd, 'test/billing.test.ts'), content: billing.files.tests },
|
||
});
|
||
}
|
||
const command = (row: typeof fixture.rows[number]) => (row.transcript[1] as any).message.content[0].input.command as string;
|
||
|
||
describe('coverage reads with neighboring display commands', () => {
|
||
test('both exact failed AQ attempts delivered source and tests in their acknowledged Bash result', () => {
|
||
expect(fixture.provenance.actualPassedCases).toBe(0);
|
||
for (const row of fixture.rows) expect(replay(row)).toEqual({ sourceRead: true, testsRead: true });
|
||
});
|
||
|
||
test('a literal grep range and numeric Git log count do not own the delivered file bytes', () => {
|
||
const first = fixture.rows[0]!, second = fixture.rows[1]!;
|
||
expect(replay(first, command(first).replace('head -40', 'head -25'))).toEqual({ sourceRead: true, testsRead: true });
|
||
expect(replay(second, command(second).replace('log --oneline -3', 'log --oneline -12'))).toEqual({ sourceRead: true, testsRead: true });
|
||
});
|
||
|
||
test.each([
|
||
['awk action', (s: string) => s.replace("awk '/^### 3\\. Test review/,/^### 4\\./'", "awk 'BEGIN { system(\"cat fake\") }'")],
|
||
['awk output redirection', (s: string) => s.replace("awk '/^### 3\\. Test review/,/^### 4\\./'", "awk '/x/ { print > \"src/billing.ts\" }'")],
|
||
['shell substitution', (s: string) => s.replace('grep -n', 'grep -n "$(cat fake)"')],
|
||
['backtick execution', (s: string) => s.replace('grep -n', 'grep -n `cat fake`')],
|
||
['quoted injected command', (s: string) => s.replace('grep -n', 'grep -n "x"; printf fake; grep -n')],
|
||
['read hidden in a conditional', (s: string) => s.replace('cat -n src/billing.ts', 'false && cat -n src/billing.ts')],
|
||
['source-only filename', (s: string) => s.replace('cat -n src/billing.ts', "echo 'cat -n src/billing.ts'")],
|
||
] as const)('%s cannot borrow source read evidence', (_, mutate) => {
|
||
const row = fixture.rows[0]!;
|
||
expect(mutate(command(row))).not.toBe(command(row));
|
||
expect(replay(row, mutate(command(row))).sourceRead).toBe(false);
|
||
});
|
||
|
||
test.each([
|
||
'git log --output=src/billing.ts -3',
|
||
'git log --ext-diff -3',
|
||
'git log --format=%x00 -3',
|
||
'git log -3; printf fake',
|
||
])('unsupported Git command %s cannot borrow delivery', git => {
|
||
const row = fixture.rows[1]!;
|
||
expect(replay(row, command(row).replace('git log --oneline -3', git)).sourceRead).toBe(false);
|
||
});
|
||
|
||
test.each(['-f/tmp/other.awk', "'-f/tmp/other.awk'", "'--source=BEGIN {print \"fake\"}'"])(
|
||
'awk input %s cannot introduce another program', operand => {
|
||
const row = fixture.rows[0]!;
|
||
const changed = command(row).replace("Test review/,/^### 4\\./' plan-eng-review/sections/review-sections.md", "Test review/,/^### 4\\./' " + operand);
|
||
expect(changed).not.toBe(command(row));
|
||
expect(replay(row, changed).sourceRead).toBe(false);
|
||
});
|
||
|
||
test('successful command identity still requires the complete file and paired parent result', () => {
|
||
for (const row of fixture.rows) {
|
||
const missing = structuredClone(row) as any;
|
||
missing.transcript[2].message.content[0].content = 'src/billing.ts and test/billing.test.ts were read';
|
||
expect(replay(missing)).toEqual({ sourceRead: false, testsRead: false });
|
||
const failed = structuredClone(row) as any;
|
||
failed.transcript[2].message.content[0].is_error = true;
|
||
expect(replay(failed)).toEqual({ sourceRead: false, testsRead: false });
|
||
}
|
||
});
|
||
});
|
||
});
|