Files
gstack/test/review-enum-lifecycle.test.ts
T
Garry Tan dcaea52800 v1.91.7.0 feat: add functional QA and pre-publication docs checks (#2983)
* feat: add surface-aware exploratory QA and ship documentation gates

* test: preserve delegated QA setup authority after main integration

* fix(qa): clarify exploration order and preserve report artifacts

* test(qa): follow the shared setup reference directly

* refactor(ship): make verification and recovery routes explicit

* test(ship): align evidence and review guards with explicit routes

* fix(workflows): clarify ship recovery and functional QA evidence

* fix(workflows): clarify approval recovery and full QA coverage

* refactor(workflows): order review transactions and clarify ship state

* fix(ship): clarify final verification and fail closed at publication

* fix(evals): attribute native atomic documentation writes

* fix(ship): clarify recovery and documentation lifecycle guidance

* fix(test): preserve observed native placeholder styling in CI

* fix(codex): report watchdog timeouts without a process-exit race

* Checkpoint functional QA implementation and workflow validation repairs

* Fix documentation and shared-review fixture contracts

* docs: clarify judge reuse and evaluation supervision

* test: align review evidence and selected case contracts

* test: verify append-only documentation checkpoints and recovery

* fix: qualify QA workflows and CI validation repairs

* fix: launch shared-libs fixture scripts on Windows

* fix: qualify QA deadlines, fixture isolation, and shard cleanup

* fix: preserve qualified QA and cancellation repairs

* fix: enforce functional fixture authority and share strict event decoding

* fix: retain free-test evidence and explain recovery

* fix: reject malformed native evidence after decoder consolidation

* test: use reliable capture for telemetry privacy filters

* test: refresh measured quick coverage and document validation costs

* Fix native fixture receipts and preserve VM validation evidence

* Align negative judge controls with upstream clarity policy

* Fix report-only QA preparation and public evidence handling

* Clarify QA-only preparation and current-report preservation

* Stream Ship quality judgments with an explicit 64k response contract

* Validate compact judge reasoning locally with supported wire schema

* Align functional QA fixture instructions with evidence acceptance

* Bind native browser diagnostics to execution evidence and align review verdicts

* Preserve native diagnostic line boundaries

* Serialize functional QA evidence from native captures

* Keep large QA evidence fixture payload out of Windows argv
2026-09-29 06:07:35 -07:00

95 lines
7.2 KiB
TypeScript

import { expect, test } from 'bun:test';
import * as fs from 'node:fs';
import * as path from 'node:path';
import { JUDGE_MS, CAPTURE_MS } from './helpers/eval-budgets';
import { SESSION_DRAIN_GRACE_MS } from './helpers/session-runner';
import { E2E_TOUCHFILES } from './helpers/touchfiles';
const source=fs.readFileSync(path.join(import.meta.dir,'skill-e2e-review.test.ts'),'utf8');
async function exercise(scenarios: Array<'success'|'timeout'|'wrong-report'|'browse-error'|'no-report'>) {
const setups:any[]=[],done:any[]=[],callbacks:any[]=[],rows:any[]=[],calls:any[]=[],outputs:string[]=[],preRunReports:string[][]=[];
const files=new Map<string,string>(); let outer=0,index=0;
const reportKeys=()=>[...files.keys()].filter(k=>/review-output-\d+\.md$/.test(k)).map(k=>k.split('/').pop()!).sort();
const args: Record<string,any>={
expect,JUDGE_MS,CAPTURE_MS,SESSION_DRAIN_GRACE_MS,ROOT:'/source',runId:'synthetic-run',process:{pid:123,env:{EVALS_RUN_ID:'synthetic-controller'}},
beforeAll:(fn:any)=>setups.push(fn),afterAll:(fn:any)=>done.push(fn),
describeIfSelected:(_title:string,names:string[],fn:any)=>{if(names.includes('review-enum-completeness'))fn();},
testConcurrentIfSelected:(name:string,fn:any,timeout:number)=>{expect(name).toBe('review-enum-completeness');callbacks.push(fn);outer=timeout;},
createEvalCollector:()=>({}),finalizeEvalCollector:()=>{},logCost:()=>{},spawnSync:()=>({status:0}),path:path.posix,os:{tmpdir:()=>'/tmp'},
fs:{mkdtempSync:(p:string)=>p+'owned',writeFileSync:(p:string,s:string)=>files.set(p,s),copyFileSync:()=>{},rmSync:(p:string)=>{files.delete(p);},
readdirSync:(p:string)=>[...files.keys()].filter(k=>k.startsWith(p+'/')).map(k=>k.slice(p.length+1)),
existsSync:(p:string)=>files.has(p),readFileSync:(p:string)=>p.startsWith('/source/')?'synthetic fixture bytes':files.get(p)},
extractSkillSections:()=> 'Review instructions',REVIEW_E2E_SECTIONS:[],
runSkillTest:async(opts:any)=>{
calls.push(opts);const scenario=scenarios[index++];
expect(opts.timeout).toBe(JUDGE_MS);expect(opts.maxTurns).toBe(15);expect(opts.model).toBeUndefined();
expect(opts.prompt).toContain('check if all consumers handle it');expect(opts.prompt).toContain('git diff main...HEAD');
expect(opts.prompt).toContain('focused, read-only core review');
expect(opts.prompt).toContain('Do not run the full /review lifecycle, QA or exploratory probes');
expect(opts.prompt).toContain('only static Ruby source with no configured runnable application, dependencies or runtime/test harness');
expect(opts.prompt).toContain('grep the sibling status values through the actual authored source and read every match in full, including unchanged consumers');
expect(opts.prompt).toContain('Do not re-run the review, reuse a prior report, or invent runtime checks');
preRunReports.push(reportKeys());
const out=opts.prompt.match(/Write your review findings once to (\S+)/)[1];outputs.push(out);
if(scenario!=='no-report')files.set(out,scenario==='wrong-report'?'Nothing to discuss.':'The returned status is missing enum handlers.');
return {exitReason:scenario==='timeout'?'timeout':'success',browseErrors:scenario==='browse-error'?['existing browser failure']:[],output:'public response'};
},
recordE2E:(_collector:any,name:string,title:string,result:any,extra:any)=>rows.push({name,title,passed:result.exitReason==='success'&&result.browseErrors.length===0,...extra,result}),
};
let body=source;for(const m of source.matchAll(/^import[\s\S]*?;\n/gm))body=body.replace(m[0],'');
new Function(...Object.keys(args),new Bun.Transpiler({loader:'ts'}).transformSync(body))(...Object.values(args));
expect(callbacks).toHaveLength(1);for(const setup of setups)await setup();
const errors:any[]=[];for(const _ of scenarios){try{await callbacks[0]();errors.push(undefined);}catch(error){errors.push(error);}}
for(const finalizer of done)await finalizer();
return {rows,calls,errors,outer,outputs,preRunReports,reportKeys:reportKeys()};
}
test('Enum caller reserves cleanup time without extending the model execution budget', async()=>{
const x=await exercise(['success']);expect(x.outer).toBe(JUDGE_MS+SESSION_DRAIN_GRACE_MS+5000);expect(x.errors).toEqual([undefined]);expect(x.rows.map(r=>r.passed)).toEqual([true]);
expect(x.reportKeys).toEqual(['review-output-1.md']);
});
test('Enum retries keep distinct capture identities and public diagnostics',async()=>{
const x=await exercise(['timeout','success']);expect(x.errors[0]).toBeDefined();expect(x.errors[1]).toBeUndefined();
expect(x.rows.map(r=>r.passed)).toEqual([false,true]);expect(new Set(x.calls.map(c=>c.runId)).size).toBe(2);
for(const c of x.calls){expect(c.runId).toStartWith('synthetic-controller-review-enum-123-');expect(c.testName).toBe('review-enum-completeness');expect(c.publicStreamDiagnostics).toBe(true);}
expect(new Set(x.outputs).size).toBe(2);
for(const out of x.outputs)expect(out).toMatch(/review-output-\d+\.md$/);
});
test('Enum retry starts with the prior report removed while its write stays captured',async()=>{
const x=await exercise(['success','success']);
expect(x.preRunReports).toEqual([[],[]]);
expect(x.reportKeys).toEqual(['review-output-2.md']);
expect(new Set(x.outputs).size).toBe(2);
});
test('Enum missing current report fails the mandatory existence assertion',async()=>{
const x=await exercise(['no-report']);expect(x.errors[0]).toBeDefined();expect(x.rows).toHaveLength(1);expect(x.rows[0].passed).toBe(false);
});
test('Enum prompt scopes a focused read-only core enum review with real fixture boundaries and a clean finish',async()=>{
const x=await exercise(['success']);const prompt=x.calls[0].prompt;
expect(prompt).toContain('focused, read-only core review');
expect(prompt).toContain('run only the checklist');
expect(prompt).toContain('Enum & Value Completeness');
expect(prompt).toContain('Do not run the full /review lifecycle, QA or exploratory probes (for example Step 4.7), Greptile, hosting/PR/review-log setup');
expect(prompt).toContain('only static Ruby source with no configured runnable application, dependencies or runtime/test harness; base main is local and there is no remote or PR');
expect(prompt).toContain('any tools that happen to be installed on the host do not expand this scope');
expect(prompt).toContain('grep the sibling status values through the actual authored source and read every match in full, including unchanged consumers');
expect(prompt).toContain('stop with a brief final response');
expect(prompt).toContain('Do not re-run the review, reuse a prior report, or invent runtime checks');
});
test('Enum semantic failure records false exactly once after the existing assertion',async()=>{
const x=await exercise(['wrong-report']);expect(x.errors[0]).toBeDefined();expect(x.rows).toHaveLength(1);expect(x.rows[0].passed).toBe(false);
});
test('Enum verdict retains the existing browser-error guard',async()=>{
const x=await exercise(['browse-error']);expect(x.rows).toHaveLength(1);expect(x.rows[0].passed).toBe(false);
});
test('Enum lifecycle controls select the existing enum owner only',()=>{
expect(Object.entries(E2E_TOUCHFILES).filter(([,paths])=>paths.includes('test/review-enum-lifecycle.test.ts')).map(([name])=>name)).toEqual(['review-enum-completeness']);
});