v1.91.7.0 feat: add functional QA and pre-publication docs checks (#2983)

* feat: add surface-aware exploratory QA and ship documentation gates

* test: preserve delegated QA setup authority after main integration

* fix(qa): clarify exploration order and preserve report artifacts

* test(qa): follow the shared setup reference directly

* refactor(ship): make verification and recovery routes explicit

* test(ship): align evidence and review guards with explicit routes

* fix(workflows): clarify ship recovery and functional QA evidence

* fix(workflows): clarify approval recovery and full QA coverage

* refactor(workflows): order review transactions and clarify ship state

* fix(ship): clarify final verification and fail closed at publication

* fix(evals): attribute native atomic documentation writes

* fix(ship): clarify recovery and documentation lifecycle guidance

* fix(test): preserve observed native placeholder styling in CI

* fix(codex): report watchdog timeouts without a process-exit race

* Checkpoint functional QA implementation and workflow validation repairs

* Fix documentation and shared-review fixture contracts

* docs: clarify judge reuse and evaluation supervision

* test: align review evidence and selected case contracts

* test: verify append-only documentation checkpoints and recovery

* fix: qualify QA workflows and CI validation repairs

* fix: launch shared-libs fixture scripts on Windows

* fix: qualify QA deadlines, fixture isolation, and shard cleanup

* fix: preserve qualified QA and cancellation repairs

* fix: enforce functional fixture authority and share strict event decoding

* fix: retain free-test evidence and explain recovery

* fix: reject malformed native evidence after decoder consolidation

* test: use reliable capture for telemetry privacy filters

* test: refresh measured quick coverage and document validation costs

* Fix native fixture receipts and preserve VM validation evidence

* Align negative judge controls with upstream clarity policy

* Fix report-only QA preparation and public evidence handling

* Clarify QA-only preparation and current-report preservation

* Stream Ship quality judgments with an explicit 64k response contract

* Validate compact judge reasoning locally with supported wire schema

* Align functional QA fixture instructions with evidence acceptance

* Bind native browser diagnostics to execution evidence and align review verdicts

* Preserve native diagnostic line boundaries

* Serialize functional QA evidence from native captures

* Keep large QA evidence fixture payload out of Windows argv
This commit is contained in:
Garry Tan authored and GitHub committed 2026-09-29 06:07:35 -07:00
1 parent 65bfb0ce49
commit dcaea52800
333 files changed
+41755 -7357

No files matched your search

+15 -10
View File
@@ -154,7 +154,6 @@ describe('selectTests', () => {
['plan-eng-review/sections/review-sections.md', 'TEST_COVERAGE_AUDIT_PLAN'],
['ship/sections/tests.md', 'TEST_BOOTSTRAP'],
['ship/sections/test-coverage.md', 'TEST_COVERAGE_AUDIT_SHIP'],
['qa/sections/test-bootstrap.md', 'TEST_BOOTSTRAP'],
['design-review/SKILL.md', 'TEST_BOOTSTRAP'],
];
for (const [output, token] of consumers) {
@@ -171,7 +170,7 @@ describe('selectTests', () => {
expect(actual.reason).toBe('diff');
expect(actual.selected.sort()).toEqual(expected);
for (const id of ['plan-eng-finding-count', 'plan-eng-multi-finding-batching',
'autoplan-chain-pty', 'plan-eng-review-format-coverage', 'ship-section-loading', 'qa-fix-loop']) {
'autoplan-chain-pty', 'plan-eng-review-format-coverage', 'ship-section-loading']) {
expect(actual.selected).toContain(id);
expect(E2E_TIERS[id]).toBe('periodic');
}
@@ -179,7 +178,7 @@ describe('selectTests', () => {
expect(actual.selected).toContain(id);
expect(E2E_TIERS[id]).toBe('gate');
}
for (const unrelated of ['browse-basic', 'retro', 'office-hours-section-loading', 'review-coverage-audit']) {
for (const unrelated of ['browse-basic', 'retro', 'office-hours-section-loading', 'review-coverage-audit', 'qa-fix-loop', 'qa-quick']) {
expect(actual.selected).not.toContain(unrelated);
}
});
@@ -199,6 +198,12 @@ describe('selectTests', () => {
expect(result.selected.sort()).toEqual(['plan-eng-review/SKILL.md sections', 'ship/SKILL.md workflow']);
});
test('ship controller guards select their workflow judge', () => {
const result = selectTests(['test/ship-control-flow.test.ts'], LLM_JUDGE_TOUCHFILES);
expect(result.reason).toBe('diff');
expect(result.selected).toEqual(['ship/SKILL.md workflow']);
});
test('bounded shared-code planning selects its consumed resolvers, excluding other Eng sections', () => {
const entrypoint = fs.readFileSync(path.join(ROOT, 'plan-eng-review/SKILL.md'), 'utf8');
const review = fs.readFileSync(path.join(ROOT, 'plan-eng-review/sections/review-sections.md'), 'utf8');
@@ -330,7 +335,7 @@ describe('selectTests', () => {
const pathCases = ['shared-libs-review-path-eligibility', 'shared-libs-review-index-flags',
'shared-libs-review-prior-coverage'];
expect(selectTests(['test/shared-libs-revalidation-prompt.test.ts'], E2E_TOUCHFILES).selected.sort())
.toEqual([...pathCases, 'shared-libs-review-revalidation'].sort());
.toEqual([...pathCases, 'shared-libs-review-revalidation', 'shared-libs-review-lifecycle'].sort());
expect(selectTests(['test/fixtures/shared-libs-index-flags-skip-question.json'], E2E_TOUCHFILES).selected.sort())
.toEqual([...pathCases, 'shared-libs-review-revalidation', 'shared-libs-review-lifecycle'].sort());
expect(selectTests(['test/fixtures/shared-libs-paths-max-turns-public.json'], E2E_TOUCHFILES).selected)
@@ -457,10 +462,10 @@ describe('selectTests', () => {
test('works with LLM_JUDGE_TOUCHFILES', () => {
const result = selectTests(['qa/SKILL.md'], LLM_JUDGE_TOUCHFILES);
expect(result.selected).toContain('qa/SKILL.md workflow');
expect(result.selected).toContain('qa/SKILL.md health rubric');
expect(result.selected).toContain('qa/SKILL.md anti-refusal');
expect(result.selected.length).toBe(3);
expect(result.selected.sort()).toEqual([
'qa/SKILL.md workflow', 'qa/SKILL.md health rubric', 'qa/SKILL.md anti-refusal',
'qa-only/SKILL.md workflow', 'review/SKILL.md workflow', 'ship/SKILL.md workflow',
].sort());
});
test('SKILL.md.tmpl root template selects root-dependent tests and routing tests', () => {
@@ -604,7 +609,7 @@ describe('TOUCHFILES completeness', () => {
);
const unique = registeredJudgeTestNames(llmContent);
expect(unique).toHaveLength(27);
expect(unique).toHaveLength(28);
const missing = unique.filter(name => !(name in LLM_JUDGE_TOUCHFILES));
if (missing.length > 0) {
@@ -623,7 +628,7 @@ describe('TOUCHFILES completeness', () => {
testIfSelected('unmapped judge case', async () => {}, 120_000);
`;
const names = registeredJudgeTestNames(withUnmappedCase);
expect(names).toHaveLength(28);
expect(names).toHaveLength(29);
expect(names.filter(name => !(name in LLM_JUDGE_TOUCHFILES))).toEqual(['unmapped judge case']);
});