Files
gstack/test/qa-caller-freshness-order.test.ts
T
Garry Tan dcaea52800 v1.91.7.0 feat: add functional QA and pre-publication docs checks (#2983)
* feat: add surface-aware exploratory QA and ship documentation gates

* test: preserve delegated QA setup authority after main integration

* fix(qa): clarify exploration order and preserve report artifacts

* test(qa): follow the shared setup reference directly

* refactor(ship): make verification and recovery routes explicit

* test(ship): align evidence and review guards with explicit routes

* fix(workflows): clarify ship recovery and functional QA evidence

* fix(workflows): clarify approval recovery and full QA coverage

* refactor(workflows): order review transactions and clarify ship state

* fix(ship): clarify final verification and fail closed at publication

* fix(evals): attribute native atomic documentation writes

* fix(ship): clarify recovery and documentation lifecycle guidance

* fix(test): preserve observed native placeholder styling in CI

* fix(codex): report watchdog timeouts without a process-exit race

* Checkpoint functional QA implementation and workflow validation repairs

* Fix documentation and shared-review fixture contracts

* docs: clarify judge reuse and evaluation supervision

* test: align review evidence and selected case contracts

* test: verify append-only documentation checkpoints and recovery

* fix: qualify QA workflows and CI validation repairs

* fix: launch shared-libs fixture scripts on Windows

* fix: qualify QA deadlines, fixture isolation, and shard cleanup

* fix: preserve qualified QA and cancellation repairs

* fix: enforce functional fixture authority and share strict event decoding

* fix: retain free-test evidence and explain recovery

* fix: reject malformed native evidence after decoder consolidation

* test: use reliable capture for telemetry privacy filters

* test: refresh measured quick coverage and document validation costs

* Fix native fixture receipts and preserve VM validation evidence

* Align negative judge controls with upstream clarity policy

* Fix report-only QA preparation and public evidence handling

* Clarify QA-only preparation and current-report preservation

* Stream Ship quality judgments with an explicit 64k response contract

* Validate compact judge reasoning locally with supported wire schema

* Align functional QA fixture instructions with evidence acceptance

* Bind native browser diagnostics to execution evidence and align review verdicts

* Preserve native diagnostic line boundaries

* Serialize functional QA evidence from native captures

* Keep large QA evidence fixture payload out of Windows argv
2026-09-29 06:07:35 -07:00

106 lines
6.5 KiB
TypeScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
import { describe, expect, test } from 'bun:test';
import { readFileSync } from 'node:fs';
import { join } from 'node:path';
import { ALL_HOST_CONFIGS } from '../hosts';
import { RESOLVERS } from '../scripts/resolvers';
import { HOST_PATHS, type TemplateContext } from '../scripts/resolvers/types';
const root = join(import.meta.dir, '..');
const compact = (text: string) => text.replace(/\s+/g, ' ');
function render(file: string, ctx: TemplateContext): string {
let text = readFileSync(join(root, file), 'utf8');
for (let pass = 0; pass < 10; pass++) {
const next = text.replace(/\{\{([A-Z_]+)(?::([^}]+))?\}\}/g, (_match, name, args) => {
if (!RESOLVERS[name]) throw new Error(`Unknown resolver ${name}`);
return RESOLVERS[name](ctx, args?.split(':'));
});
if (next === text) return compact(text);
text = next;
}
throw new Error(`Unresolved template ${file}`);
}
function ordered(text: string, markers: string[]): void {
let previous = -1;
for (const marker of markers) {
const index = text.indexOf(marker);
expect(index, marker).toBeGreaterThan(previous);
previous = index;
}
}
describe('review and ship completion freshness contracts', () => {
for (const host of ALL_HOST_CONFIGS) {
for (const skillName of ['review', 'ship']) {
const ctx: TemplateContext = { host: host.name, skillName, tmplPath: '', paths: HOST_PATHS[host.name] };
const body = compact(RESOLVERS.QA_REVIEW(ctx));
const shared = compact(RESOLVERS.QA_EXPLORATORY({ ...ctx, skillName: 'qa' }));
const gate = body.slice(body.indexOf('**4. Check freshness before reporting.**'), body.indexOf('Return verified defects'));
test(`${host.name}/${skillName}: dependent probes await prerequisites without serializing independent Reads`, () => {
expect(shared).toContain('Complete these Reads in order before writing charters or probing');
expect(shared).toContain('Wait for successful checkpoint publication before dispatch');
expect(body).toContain('Await clock/guard results before acting');
expect(body).toContain('Batch only independent Reads');
expect(shared).toContain('Missing or unreadable assets, prerequisites or permission block affected probes, not independent safe checks');
ordered(body, ['Batch only independent Reads', '**1.', '**3. Run smoke and plan checks']);
});
test(`${host.name}/${skillName}: normal and skipped paths resolve freshness before completion`, () => {
expect(gate).toContain('Before every completion report or log');
expect(gate).toContain('even with zero fixes or skipped specialists');
ordered(gate, ['a. Read agent/user updates', 'await results without batching them with reporting/logging',
"b. Compare each probe's recorded", 'c. Re-review', 'd. Compare again after revalidation',
'Report clean/completed only when all required checks pass on current inputs']);
expect(gate).toContain('even without updates');
});
test(`${host.name}/${skillName}: late changes preserve current per-probe evidence`, () => {
expect(gate).toContain("Compare each probe's recorded source, tests, contracts, commands and fixtures (or input fingerprint) with current inputs");
expect(gate).toContain('Never rerun valid current passes');
expect(gate).toContain('Re-review changed or uncertain coverage and repeat step 3 for affected checks');
expect(gate).toContain('Compare again after revalidation or edits/updates');
});
test(`${host.name}/${skillName}: required revalidation uses real limits rather than the optional-work reserve`, () => {
expect(gate).toContain('repeat step 3 for affected checks');
expect(shared).toContain('return to step 2 for each affected revalidation');
expect(shared).toContain('Keep limits/notes; status requires fresh evidence');
expect(gate).toContain('Reporting reserves cannot stop required revalidation within the caller\'s deadline');
expect(body).toContain('Await clock/guard results before acting');
expect(body).toContain('Smoke: 5 minutes/12 probes');
expect(body).toContain('Then run required plan checks, even after smoke expires');
expect(body).toContain('no smoke guard; never reset the clock');
expect(body).toContain('Use finite command timeouts, capped at the caller\'s remaining time if it has a deadline');
});
test(`${host.name}/${skillName}: unavailable freshness or insufficient time cannot certify completion`, () => {
expect(gate).toContain('Failed or unavailable Reads or insufficient time block affected required checks');
expect(gate).toContain('List failed, blocked, inconclusive and not-run checks');
expect(gate).toContain('Report clean/completed only when all required checks pass on current inputs; optional untested ideas do not block it');
expect(body).toContain('Only the parent runs report-only discovery');
expect(body).toContain('Test creation needs user approval');
expect(body).toContain('Setup/permission blockers are not defects');
expect(body).toContain(skillName === 'review'
? 'Unresolved coverage makes Step 5.8 incomplete; a ship waiver cannot complete it'
: 'explicit named-risk acceptance; otherwise blocked');
});
}
test(`${host.name}/ship: finalization consumes the freshness result before summaries and persistence`, () => {
const ctx: TemplateContext = { host: host.name, skillName: 'ship', tmplPath: '', paths: HOST_PATHS[host.name] };
const body = render('ship/sections/review-army.md.tmpl', ctx);
const finalization = body.slice(body.indexOf('4. **Finish and log'), body.indexOf('5. Output summary:'));
expect(finalization).toContain('Recheck freshness (Step 9.2.1) before items 5–6');
expect(body).toContain('even with zero fixes or skipped specialists');
expect(body).toContain('Report clean/completed only when all required checks pass on current inputs');
ordered(body, ['Recheck freshness (Step 9.2.1) before items 5–6', '5. Output summary:', '6. Persist the review result']);
expect(body).toContain('Complete items 5–6 exactly once with the original REVIEW_START');
expect(body).toContain('Missing dispatched output uses `status:"unavailable"`, `completed:false` and `converged:false`');
expect(body).toContain('Failed, blocked, inconclusive or not-run required probes mean false, never clean');
expect(body).toContain('Undispatched host-unsupported/gated specialists do not block');
});
}
});