mirror of
https://github.com/garrytan/gstack.git
synced 2026-10-02 17:40:02 +02:00
v1.91.7.0 feat: add functional QA and pre-publication docs checks (#2983)
* feat: add surface-aware exploratory QA and ship documentation gates * test: preserve delegated QA setup authority after main integration * fix(qa): clarify exploration order and preserve report artifacts * test(qa): follow the shared setup reference directly * refactor(ship): make verification and recovery routes explicit * test(ship): align evidence and review guards with explicit routes * fix(workflows): clarify ship recovery and functional QA evidence * fix(workflows): clarify approval recovery and full QA coverage * refactor(workflows): order review transactions and clarify ship state * fix(ship): clarify final verification and fail closed at publication * fix(evals): attribute native atomic documentation writes * fix(ship): clarify recovery and documentation lifecycle guidance * fix(test): preserve observed native placeholder styling in CI * fix(codex): report watchdog timeouts without a process-exit race * Checkpoint functional QA implementation and workflow validation repairs * Fix documentation and shared-review fixture contracts * docs: clarify judge reuse and evaluation supervision * test: align review evidence and selected case contracts * test: verify append-only documentation checkpoints and recovery * fix: qualify QA workflows and CI validation repairs * fix: launch shared-libs fixture scripts on Windows * fix: qualify QA deadlines, fixture isolation, and shard cleanup * fix: preserve qualified QA and cancellation repairs * fix: enforce functional fixture authority and share strict event decoding * fix: retain free-test evidence and explain recovery * fix: reject malformed native evidence after decoder consolidation * test: use reliable capture for telemetry privacy filters * test: refresh measured quick coverage and document validation costs * Fix native fixture receipts and preserve VM validation evidence * Align negative judge controls with upstream clarity policy * Fix report-only QA preparation and public evidence handling * Clarify QA-only preparation and current-report preservation * Stream Ship quality judgments with an explicit 64k response contract * Validate compact judge reasoning locally with supported wire schema * Align functional QA fixture instructions with evidence acceptance * Bind native browser diagnostics to execution evidence and align review verdicts * Preserve native diagnostic line boundaries * Serialize functional QA evidence from native captures * Keep large QA evidence fixture payload out of Windows argv
This commit is contained in:
1 parent
65bfb0ce49
commit
dcaea52800
333 files changed
+41755
-7357
No files matched your search
+15
-10
@@ -154,7 +154,6 @@ describe('selectTests', () => {
|
||||
['plan-eng-review/sections/review-sections.md', 'TEST_COVERAGE_AUDIT_PLAN'],
|
||||
['ship/sections/tests.md', 'TEST_BOOTSTRAP'],
|
||||
['ship/sections/test-coverage.md', 'TEST_COVERAGE_AUDIT_SHIP'],
|
||||
['qa/sections/test-bootstrap.md', 'TEST_BOOTSTRAP'],
|
||||
['design-review/SKILL.md', 'TEST_BOOTSTRAP'],
|
||||
];
|
||||
for (const [output, token] of consumers) {
|
||||
@@ -171,7 +170,7 @@ describe('selectTests', () => {
|
||||
expect(actual.reason).toBe('diff');
|
||||
expect(actual.selected.sort()).toEqual(expected);
|
||||
for (const id of ['plan-eng-finding-count', 'plan-eng-multi-finding-batching',
|
||||
'autoplan-chain-pty', 'plan-eng-review-format-coverage', 'ship-section-loading', 'qa-fix-loop']) {
|
||||
'autoplan-chain-pty', 'plan-eng-review-format-coverage', 'ship-section-loading']) {
|
||||
expect(actual.selected).toContain(id);
|
||||
expect(E2E_TIERS[id]).toBe('periodic');
|
||||
}
|
||||
@@ -179,7 +178,7 @@ describe('selectTests', () => {
|
||||
expect(actual.selected).toContain(id);
|
||||
expect(E2E_TIERS[id]).toBe('gate');
|
||||
}
|
||||
for (const unrelated of ['browse-basic', 'retro', 'office-hours-section-loading', 'review-coverage-audit']) {
|
||||
for (const unrelated of ['browse-basic', 'retro', 'office-hours-section-loading', 'review-coverage-audit', 'qa-fix-loop', 'qa-quick']) {
|
||||
expect(actual.selected).not.toContain(unrelated);
|
||||
}
|
||||
});
|
||||
@@ -199,6 +198,12 @@ describe('selectTests', () => {
|
||||
expect(result.selected.sort()).toEqual(['plan-eng-review/SKILL.md sections', 'ship/SKILL.md workflow']);
|
||||
});
|
||||
|
||||
test('ship controller guards select their workflow judge', () => {
|
||||
const result = selectTests(['test/ship-control-flow.test.ts'], LLM_JUDGE_TOUCHFILES);
|
||||
expect(result.reason).toBe('diff');
|
||||
expect(result.selected).toEqual(['ship/SKILL.md workflow']);
|
||||
});
|
||||
|
||||
test('bounded shared-code planning selects its consumed resolvers, excluding other Eng sections', () => {
|
||||
const entrypoint = fs.readFileSync(path.join(ROOT, 'plan-eng-review/SKILL.md'), 'utf8');
|
||||
const review = fs.readFileSync(path.join(ROOT, 'plan-eng-review/sections/review-sections.md'), 'utf8');
|
||||
@@ -330,7 +335,7 @@ describe('selectTests', () => {
|
||||
const pathCases = ['shared-libs-review-path-eligibility', 'shared-libs-review-index-flags',
|
||||
'shared-libs-review-prior-coverage'];
|
||||
expect(selectTests(['test/shared-libs-revalidation-prompt.test.ts'], E2E_TOUCHFILES).selected.sort())
|
||||
.toEqual([...pathCases, 'shared-libs-review-revalidation'].sort());
|
||||
.toEqual([...pathCases, 'shared-libs-review-revalidation', 'shared-libs-review-lifecycle'].sort());
|
||||
expect(selectTests(['test/fixtures/shared-libs-index-flags-skip-question.json'], E2E_TOUCHFILES).selected.sort())
|
||||
.toEqual([...pathCases, 'shared-libs-review-revalidation', 'shared-libs-review-lifecycle'].sort());
|
||||
expect(selectTests(['test/fixtures/shared-libs-paths-max-turns-public.json'], E2E_TOUCHFILES).selected)
|
||||
@@ -457,10 +462,10 @@ describe('selectTests', () => {
|
||||
|
||||
test('works with LLM_JUDGE_TOUCHFILES', () => {
|
||||
const result = selectTests(['qa/SKILL.md'], LLM_JUDGE_TOUCHFILES);
|
||||
expect(result.selected).toContain('qa/SKILL.md workflow');
|
||||
expect(result.selected).toContain('qa/SKILL.md health rubric');
|
||||
expect(result.selected).toContain('qa/SKILL.md anti-refusal');
|
||||
expect(result.selected.length).toBe(3);
|
||||
expect(result.selected.sort()).toEqual([
|
||||
'qa/SKILL.md workflow', 'qa/SKILL.md health rubric', 'qa/SKILL.md anti-refusal',
|
||||
'qa-only/SKILL.md workflow', 'review/SKILL.md workflow', 'ship/SKILL.md workflow',
|
||||
].sort());
|
||||
});
|
||||
|
||||
test('SKILL.md.tmpl root template selects root-dependent tests and routing tests', () => {
|
||||
@@ -604,7 +609,7 @@ describe('TOUCHFILES completeness', () => {
|
||||
);
|
||||
|
||||
const unique = registeredJudgeTestNames(llmContent);
|
||||
expect(unique).toHaveLength(27);
|
||||
expect(unique).toHaveLength(28);
|
||||
|
||||
const missing = unique.filter(name => !(name in LLM_JUDGE_TOUCHFILES));
|
||||
if (missing.length > 0) {
|
||||
@@ -623,7 +628,7 @@ describe('TOUCHFILES completeness', () => {
|
||||
testIfSelected('unmapped judge case', async () => {}, 120_000);
|
||||
`;
|
||||
const names = registeredJudgeTestNames(withUnmappedCase);
|
||||
expect(names).toHaveLength(28);
|
||||
expect(names).toHaveLength(29);
|
||||
expect(names.filter(name => !(name in LLM_JUDGE_TOUCHFILES))).toEqual(['unmapped judge case']);
|
||||
});
|
||||
|
||||
|
||||
Reference in new issue
Block a user