v1.91.7.0 feat: add functional QA and pre-publication docs checks (#2983)

* feat: add surface-aware exploratory QA and ship documentation gates

* test: preserve delegated QA setup authority after main integration

* fix(qa): clarify exploration order and preserve report artifacts

* test(qa): follow the shared setup reference directly

* refactor(ship): make verification and recovery routes explicit

* test(ship): align evidence and review guards with explicit routes

* fix(workflows): clarify ship recovery and functional QA evidence

* fix(workflows): clarify approval recovery and full QA coverage

* refactor(workflows): order review transactions and clarify ship state

* fix(ship): clarify final verification and fail closed at publication

* fix(evals): attribute native atomic documentation writes

* fix(ship): clarify recovery and documentation lifecycle guidance

* fix(test): preserve observed native placeholder styling in CI

* fix(codex): report watchdog timeouts without a process-exit race

* Checkpoint functional QA implementation and workflow validation repairs

* Fix documentation and shared-review fixture contracts

* docs: clarify judge reuse and evaluation supervision

* test: align review evidence and selected case contracts

* test: verify append-only documentation checkpoints and recovery

* fix: qualify QA workflows and CI validation repairs

* fix: launch shared-libs fixture scripts on Windows

* fix: qualify QA deadlines, fixture isolation, and shard cleanup

* fix: preserve qualified QA and cancellation repairs

* fix: enforce functional fixture authority and share strict event decoding

* fix: retain free-test evidence and explain recovery

* fix: reject malformed native evidence after decoder consolidation

* test: use reliable capture for telemetry privacy filters

* test: refresh measured quick coverage and document validation costs

* Fix native fixture receipts and preserve VM validation evidence

* Align negative judge controls with upstream clarity policy

* Fix report-only QA preparation and public evidence handling

* Clarify QA-only preparation and current-report preservation

* Stream Ship quality judgments with an explicit 64k response contract

* Validate compact judge reasoning locally with supported wire schema

* Align functional QA fixture instructions with evidence acceptance

* Bind native browser diagnostics to execution evidence and align review verdicts

* Preserve native diagnostic line boundaries

* Serialize functional QA evidence from native captures

* Keep large QA evidence fixture payload out of Windows argv
This commit is contained in:
Garry Tan authored and GitHub committed 2026-09-29 06:07:35 -07:00
1 parent 65bfb0ce49
commit dcaea52800
333 files changed
+41755 -7357

No files matched your search

+31 -9
View File
@@ -6,10 +6,13 @@ import { CAPTURE_LONG_MS } from './helpers/eval-budgets';
import { describeE2ETier, e2eTierEnabled } from './helpers/e2e-gate';
import { EvalCollector } from './helpers/eval-store';
import {
fixtureWorkingTree, reviewLifecycleInstructions, reviewRevalidationPrompt, reviewRecords,
fixtureGit, fixtureWorkingTree, reviewLifecycleInstructions, reviewRevalidationPrompt, reviewRecords, SHARED_LIBS_ROOT,
runSharedInteractive, toolCommandTrace, readRequests, SharedCaptureAccumulator, type SharedLibsFixture,
} from './helpers/shared-libs-eval-fixture';
import { preparePathEligibilityFixture, type PathEligibilityCase } from './helpers/shared-libs-path-fixture';
import {
preparePathEligibilityFixture, checkPathReviewPrerequisites, hasPathReviewPrerequisiteReceipt, type PathEligibilityCase,
} from './helpers/shared-libs-path-fixture';
import { hasTrustedSharedLibsCheck } from './helpers/shared-libs-review-start-evidence';
const describeE2E = describeE2ETier('gate');
const collector = e2eTierEnabled('gate') ? new EvalCollector('e2e') : null;
@@ -51,7 +54,9 @@ function sourceReadTrace(result: any, fixture: SharedLibsFixture, sources: strin
if (resolved === path.resolve(repo, source)) continue;
const contents = fs.readFileSync(resolved, 'utf8');
if (!contents) continue;
const spellings = [relative, resolved].map(value => value.replace(/[.*+?^${}()|[\]\\]/g, '\\$&'));
const paths = [relative, resolved];
if (path.sep === '\\') paths.push(...paths.map(value => value.replaceAll('\\', '/')));
const spellings = paths.map(value => value.replace(/[.*+?^${}()|[\]\\]/g, '\\$&'));
const namedPath = new RegExp(`(?:^|[\\s"'=;])(?:\\./)?(?:${spellings.join('|')})(?=$|[\\s"';|)])`);
if (returned.some(({ tool, read, text }) => (tool === 'Read' ? path.resolve(repo, read) === resolved : namedPath.test(read))
&& text.includes(contents))) reads.push(source);
@@ -72,21 +77,25 @@ async function exerciseEligibility(testId: string, kinds: PathEligibilityCase[])
try {
const prepared = preparePathEligibilityFixture(kind);
f = prepared.fixture;
const prerequisites = checkPathReviewPrerequisites(f, prepared.resumed.input);
expect(prerequisites.settled, `${kind}: fixture prerequisites must be settled before capture`).toBe(true);
const sourceBefore = new Map(prepared.current.evidence_paths.map((source: string) =>
[source, fs.readFileSync(path.join(prepared.fixture.repo, source), 'utf8')]));
const instructions = reviewLifecycleInstructions(f);
const supplied = path.join(f.root, 'current-advisory.jsonl');
fs.writeFileSync(supplied, JSON.stringify({ ...prepared.current, specialist: 'maintainability' }) + '\n');
const prompt = reviewRevalidationPrompt(f, instructions, supplied)
const prompt = reviewRevalidationPrompt(f, instructions, supplied, prepared.resumed)
+ '\nAll named caller sources are first-party authored runtime code. Inspect them directly, including any Git/path boundary, before deciding whether the previous review decision can be reused. The fixture contains no generated caller sources.';
const capture = await runSharedInteractive(f, testId, prompt, 'skip');
const capture = await runSharedInteractive(f, testId, prompt, 'skip', { attempt });
result = capture.result;
expect(result.exitReason, `${kind}: ${result.output}`).toBe('success');
expect(result.toolCalls.length).toBeGreaterThan(0);
expect(checkPathReviewPrerequisites(f, prepared.resumed.input), `${kind}: prerequisite state must remain unchanged`).toEqual(prerequisites);
expect(hasPathReviewPrerequisiteReceipt(result.events ?? [], prepared.resumed.checkCommand, prerequisites),
`${kind}: consume current synthetic prerequisites before persistence`).toBe(true);
expect(capture.questions.length, `${kind}: old decision must be revalidated and presented again`).toBeGreaterThan(0);
const trace = toolCommandTrace(result).join('\n');
expect(trace).toContain('gstack-review-read');
expect(trace).toContain('sharedLibsFingerprint');
expect(trace).toContain('gstack-review-log');
expect(trace).toContain('--start');
expect(trace).toContain('--finish');
@@ -96,9 +105,6 @@ async function exerciseEligibility(testId: string, kinds: PathEligibilityCase[])
for (const source of prepared.sourcePaths) {
expect(reads, `${kind}: reread ${source}`).toContain(source);
}
if (kind === 'symlinks') expect(trace).toMatch(/readlink|lstat|stat\b|test\s+-L|\[\s+-L|ls-files[^\n]*(?:--stage|-s\b)/);
if (kind === 'submodule') expect(trace + '\n' + result.output).toMatch(/submodule|160000/i);
if (kind === 'ignored') expect(trace + '\n' + result.output).toMatch(/check-ignore|ignored|exclude-standard/i);
expect(fixtureWorkingTree(f), `${kind}: Skip must not refactor any source`).toBe(prepared.beforeTree);
for (const [source, before] of sourceBefore) {
expect(fs.readFileSync(path.join(f.repo, source as string), 'utf8'), `${kind}: preserve raw ${source}`).toBe(before);
@@ -112,6 +118,21 @@ async function exerciseEligibility(testId: string, kinds: PathEligibilityCase[])
expect(skipped.length, `${kind}: the new explicit decision must be saved`).toBeGreaterThan(0);
expect(skipped.some((finding: any) => finding.helper_target?.path === 'lib/retry-after.ts'
&& finding.helper_target?.symbol === 'retrySeconds')).toBe(true);
if (trace.includes('--check-shared-libs')) {
expect(hasTrustedSharedLibsCheck(result.events ?? result.transcript ?? [], {
helper: path.join(SHARED_LIBS_ROOT, 'bin/gstack-review-log'),
repo: f.repo, state: f.state, slug: 'fixture-shared-libs',
directory: path.join(f.state, 'projects/fixture-shared-libs/.review-starts'),
branch: fixtureGit(f, 'symbolic-ref', '--quiet', '--short', 'HEAD'), wtree: fixtureWorkingTree(f),
startedAt: last.review_binding.started_at, finding: prepared.current, reusable: false,
coveredPaths: skipped.find((finding: any) => finding.fingerprint === prepared.current.fingerprint)?.snapshot_covered_paths,
})).toBe(true);
} else {
expect(trace).toContain('sharedLibsFingerprint');
if (kind === 'symlinks') expect(trace).toMatch(/readlink|lstat|stat\b|test\s+-L|\[\s+-L|ls-files[^\n]*(?:--stage|-s\b)/);
if (kind === 'submodule') expect(trace + '\n' + result.output).toMatch(/submodule|160000/i);
if (kind === 'ignored') expect(trace + '\n' + result.output).toMatch(/check-ignore|ignored|exclude-standard/i);
}
for (const finding of skipped) {
expect(finding.fingerprint).toMatch(/^shared-libs:[0-9a-f]{64}$/);
expect(finding.evidence_paths).toContain('src/retry-worker.ts');
@@ -135,6 +156,7 @@ async function exerciseEligibility(testId: string, kinds: PathEligibilityCase[])
duration_ms: result?.durationMs ?? 0, cost_usd: result?.costUsd ?? 0,
model: result?.model, turns_used: result?.turnsUsed ?? 0,
transcript: [{ scenario: kind, error: scenarioError, diagnostic: captureDiagnostic,
prerequisite_source: 'synthetic-fixture-input', prerequisite_native_coverage: false,
cost_known: result?.costKnown, provider_requests: f ? readRequests(f) : [] }, ...(result?.events ?? [])],
output: `[${kind}]${scenarioError ? ` ${scenarioError}` : ''}\n${result?.output ?? ''}`,
error: scenarioError,