mirror of
https://github.com/garrytan/gstack.git
synced 2026-10-04 10:26:52 +02:00
v1.91.7.0 feat: add functional QA and pre-publication docs checks (#2983)
* feat: add surface-aware exploratory QA and ship documentation gates * test: preserve delegated QA setup authority after main integration * fix(qa): clarify exploration order and preserve report artifacts * test(qa): follow the shared setup reference directly * refactor(ship): make verification and recovery routes explicit * test(ship): align evidence and review guards with explicit routes * fix(workflows): clarify ship recovery and functional QA evidence * fix(workflows): clarify approval recovery and full QA coverage * refactor(workflows): order review transactions and clarify ship state * fix(ship): clarify final verification and fail closed at publication * fix(evals): attribute native atomic documentation writes * fix(ship): clarify recovery and documentation lifecycle guidance * fix(test): preserve observed native placeholder styling in CI * fix(codex): report watchdog timeouts without a process-exit race * Checkpoint functional QA implementation and workflow validation repairs * Fix documentation and shared-review fixture contracts * docs: clarify judge reuse and evaluation supervision * test: align review evidence and selected case contracts * test: verify append-only documentation checkpoints and recovery * fix: qualify QA workflows and CI validation repairs * fix: launch shared-libs fixture scripts on Windows * fix: qualify QA deadlines, fixture isolation, and shard cleanup * fix: preserve qualified QA and cancellation repairs * fix: enforce functional fixture authority and share strict event decoding * fix: retain free-test evidence and explain recovery * fix: reject malformed native evidence after decoder consolidation * test: use reliable capture for telemetry privacy filters * test: refresh measured quick coverage and document validation costs * Fix native fixture receipts and preserve VM validation evidence * Align negative judge controls with upstream clarity policy * Fix report-only QA preparation and public evidence handling * Clarify QA-only preparation and current-report preservation * Stream Ship quality judgments with an explicit 64k response contract * Validate compact judge reasoning locally with supported wire schema * Align functional QA fixture instructions with evidence acceptance * Bind native browser diagnostics to execution evidence and align review verdicts * Preserve native diagnostic line boundaries * Serialize functional QA evidence from native captures * Keep large QA evidence fixture payload out of Windows argv
This commit is contained in:
1 parent
65bfb0ce49
commit
dcaea52800
333 files changed
+41755
-7357
No files matched your search
@@ -0,0 +1,88 @@
|
||||
import { afterAll, expect } from 'bun:test';
|
||||
import * as fs from 'node:fs';
|
||||
import * as path from 'node:path';
|
||||
import {
|
||||
createEvalCollector, describeIfSelected, finalizeEvalCollector, logCost, recordE2E,
|
||||
testConcurrentIfSelected,
|
||||
} from './helpers/e2e-helpers';
|
||||
import { getProjectEvalDir } from './helpers/eval-store';
|
||||
import {
|
||||
createQaCallerFixture, QA_CALLER_CASES, QA_CALLER_TEST_MS, readCallerReceipt,
|
||||
retainQaCallerEvidence, runQaCaller, validateCallerEvidence,
|
||||
type QaCallerCase,
|
||||
} from './helpers/qa-callers-fixture';
|
||||
import type { SkillTestResult } from './helpers/session-runner';
|
||||
import { readQACheckpointFiles } from './helpers/qa-checkpoint-evidence';
|
||||
|
||||
const collector = createEvalCollector('e2e-qa-callers');
|
||||
let sequence = 0;
|
||||
|
||||
async function capture(caseId: QaCallerCase) {
|
||||
const run = process.env.EVALS_RUN_ID;
|
||||
if (!run || !/^[\w.-]+$/.test(run)) throw new Error('Caller captures require an explicit safe EVALS_RUN_ID for durable native evidence');
|
||||
const id = `${run}-${caseId}-${process.pid}-${++sequence}`;
|
||||
const fixture = createQaCallerFixture(caseId);
|
||||
let result: SkillTestResult | undefined;
|
||||
let passed = false;
|
||||
try {
|
||||
await fixture.observe();
|
||||
result = await runQaCaller(fixture, id);
|
||||
await fixture.close();
|
||||
const receipt = readCallerReceipt(fixture);
|
||||
const probes = fixture.probes();
|
||||
const requiredCharters = caseId === 'ship-exploratory-plan-checks' ? ['happy', 'adverse', 'plan:nine'] : ['happy', 'adverse'];
|
||||
const errors = validateCallerEvidence({
|
||||
caller: fixture.caller, result, probes, receipt,
|
||||
currentSnapshot: fixture.snapshot(), requiredCharters,
|
||||
mutations: fixture.mutationEvents, observerComplete: fixture.observation?.complete === true && fixture.observerErrors.length === 0,
|
||||
workflowCommands: fixture.workflowCommands,
|
||||
fixtureRoot: fixture.cwd,
|
||||
runtime: fixture.runtime,
|
||||
requireGuardedSmoke: true,
|
||||
requireCapturedEvidence: true,
|
||||
reportRoot: path.join(fixture.cwd, 'reports'),
|
||||
checkpointFiles: readQACheckpointFiles(path.join(fixture.cwd, 'reports')),
|
||||
reportMarkdown: fs.readFileSync(path.join(fixture.cwd, 'reports/review.md'), 'utf8'),
|
||||
});
|
||||
expect(errors).toEqual([]);
|
||||
expect(fs.readFileSync(path.join(fixture.cwd, 'reports/review.md'), 'utf8').trim().length).toBeGreaterThan(100);
|
||||
if (caseId === 'review-exploratory-small-cli') {
|
||||
const defect = probes.find(probe => probe.input === '0' && probe.status === 'fail');
|
||||
expect(defect).toBeDefined();
|
||||
expect(receipt.probes).toContain(defect!.id);
|
||||
expect(['fail', 'blocked']).toContain(receipt.status);
|
||||
} else if (caseId === 'ship-exploratory-unavailable') {
|
||||
const blocked = probes.find(probe => probe.status === 'blocked' && probe.exit !== 0);
|
||||
expect(blocked).toBeDefined();
|
||||
expect(receipt.probes).toContain(blocked!.id);
|
||||
expect(receipt.status).toBe('blocked');
|
||||
expect(receipt.remaining.length).toBeGreaterThan(0);
|
||||
} else {
|
||||
expect(receipt.status).toBe('pass');
|
||||
if (caseId === 'ship-exploratory-late-input') {
|
||||
expect(fixture.lateApplied).toBe(true);
|
||||
expect(new Set(probes.map(probe => probe.snapshot)).size).toBeGreaterThan(1);
|
||||
}
|
||||
}
|
||||
passed = true;
|
||||
} finally {
|
||||
await fixture.close();
|
||||
const artifacts = path.join(process.env.GSTACK_EVAL_DIR || getProjectEvalDir(), 'qa-callers', id);
|
||||
retainQaCallerEvidence(fixture, artifacts, result);
|
||||
if (result) {
|
||||
logCost(caseId, result);
|
||||
recordE2E(collector, caseId, 'Automatic parent exploratory QA', result, { passed });
|
||||
}
|
||||
fs.rmSync(fixture.root, { recursive: true, force: true });
|
||||
expect(fs.existsSync(path.join(artifacts, 'native-events.json'))).toBe(true);
|
||||
expect(fs.statSync(path.join(artifacts, 'native-probes.jsonl')).mode & 0o777).toBe(0o600);
|
||||
}
|
||||
}
|
||||
|
||||
describeIfSelected('Automatic parent exploratory QA', [...QA_CALLER_CASES], () => {
|
||||
for (const caseId of QA_CALLER_CASES) {
|
||||
testConcurrentIfSelected(caseId, () => capture(caseId), QA_CALLER_TEST_MS);
|
||||
}
|
||||
});
|
||||
|
||||
afterAll(() => finalizeEvalCollector(collector));
|
||||
Reference in new issue
Block a user