Merge origin/main (v1.91.7.0) into test-audit-reduction

Keep both intents: v1.91.7.0's functional QA, docsync and exploratory
paid cases and their free owners stay; this branch's deletions stay
deleted. main's new paid keys follow the derived-closure touchfile rule
(free *.test.ts paths dropped, static helper/fixture closure added), its
new helper-only tests join the ratchet baseline, and its free selection
examples that named free test files now assert the derived selection.

Periodic CI keeps seven slices without the retired Autoplan slice; the
gate census keeps seven single-worker slices with --skip-judges. Wall
and census literals are recomputed from the merged planner, durations
are re-recorded on Ubicloud, and VERSION stays 1.91.8.0 above 1.91.7.0.
This commit is contained in:
garrytan committed 2026-09-29 13:48:02 +00:00
commit b421bba2c9
325 files changed
+42569 -8257

No files matched your search

+36 -15
View File
@@ -55,7 +55,7 @@ function expectOutsideReviewControlFlow(text: string, promptHeading: string): vo
describe('workflow judge excerpts', () => {
test('expands ship sections in execution order, not alphabetical order', () => {
const text = readWorkflowExcerpt('ship/SKILL.md', '# Ship:', '## Important Rules');
const headings = ['## Step 3:', '## Step 4:', '## Step 7:', '## Step 8:', '## Step 9:', '## Step 10:', '## Step 11:', '## Step 12:', '## Step 13:', '## Step 14:'];
const headings = ['## Step 3:', '## Step 4:', '## Step 7:', '## Step 8:', '## Step 9:', '## Step 10:', '## Step 11:', '## Step 11.5:', '## Step 12:', '## Step 13:', '## Step 14:'];
const indices = headings.map(heading => text.indexOf(heading));
expect(indices.every(index => index >= 0)).toBe(true);
expect(indices).toEqual([...indices].sort((a, b) => a - b));
@@ -80,12 +80,29 @@ describe('workflow judge excerpts', () => {
test('ship review shortcuts retain dedup and fixes repeat the whole review cycle', () => {
const text = readWorkflowExcerpt('ship/SKILL.md', '# Ship:', '## Important Rules');
expect(text).toContain('Continue to Step 9.3 (cross-review dedup)');
expect(text).toContain("Continue to Step 9.2 with the core/design-lite findings and an empty specialist list, then the parent's Exploratory QA step and Step 9.3 (cross-review dedup)");
expect(text).toContain('## Step 9.4: Fix-First and persistence');
expect(text).toContain('including design, specialists, Red Team, and dedup');
expect(text.replace(/\s+/g, ' ')).toContain('Run checklist/design, specialists (9.1), merge/Red Team (9.2), exploratory QA (9.2.1), dedup (9.3), then fixes and logging (9.4)');
expect(text.replace(/\s+/g, ' ')).toContain('**Fixes applied below the cap:** Insert Step 5, affected Steps 6–8 and all of Step 9 before the pending Step 10 in the work list. Tests must pass or retain approval for the same verified pre-existing failures and scope');
const audit = text.slice(text.indexOf('## Step 7:'), text.indexOf('## Step 8:'));
expect(audit).not.toContain('Scope Challenge');
expect(text).toContain('Ship anyway retains VERIFY_RESULT=fail');
expect(text.replace(/\s+/g, ' ')).toContain('Keep actual outcomes and incomplete flags; VERIFY_RESULT stays fail for plan-check exceptions');
});
test('ship excerpt preserves readable detours, audit fallback and final input decisions', () => {
const text = readWorkflowExcerpt('ship/SKILL.md', '# Ship:', '## Important Rules').replace(/\s+/g, ' ');
expect(text).toContain('For another repair, repeat rule 2 without discarding pending work');
expect(text).toContain('A further Step 9 fix affecting 6–8 makes the list `5 → 6 → 7 → 8 → 9 → 10 → 11 → 11.5`');
expect(text).toContain('The unchanged release steps follow');
expect(text).toContain('All three snapshots must match');
expect(text).toContain('does not mean the failed or unrun probes passed');
expect(text).toContain('Fallback recovers the audit; it does not pass or bypass the coverage gate');
expect(text).toContain('Skip only the plan completion audit');
expect(text).toContain('Continue with Step 8.1, Scope Drift and Prior Learnings');
expect(text).toContain('Step 9 QA still runs');
expect(text).toContain('Use this example only after confirming that every allowed edit is release metadata');
expect(text).not.toContain('Every listed change below is metadata:');
expect(text).not.toContain('No plan file found:** Skip entirely');
});
test('a sliced section is not appended again with its generated header', () => {
@@ -109,7 +126,12 @@ describe('workflow judge excerpts', () => {
expect(text).toContain('never create an empty commit');
const review = text.slice(text.indexOf('## Step 9:'), text.indexOf('## Step 10:'));
expect(review.indexOf('## Confidence Calibration')).toBeLessThan(review.indexOf('1. Read'));
expect(review).toContain('Continue to Step 10 only after a completed, converged review is persisted');
const flat = review.replace(/\s+/g, ' ');
expect(flat).toContain('**No edits in this pass:** Resolve the required-probe gate below. Only after it clears may you continue to Step 10');
expect(flat).toContain('**Dispatched reviewer output missing:** STOP');
expect(flat).toContain('Retain queued fixes and restore coverage');
expect(flat).toContain('**Third fixing cycle reached (`CYCLES >= 3`):** STOP and report recurring findings with `converged:false`; do not run a fourth fixing cycle');
expect(flat).toContain('With completed checklist and dispatched reviewers, failed/unavailable required probes block continuation');
});
test('ship approval gates stay outside the subagent prompts', () => {
@@ -123,7 +145,7 @@ describe('workflow judge excerpts', () => {
expect(section.indexOf(gate)).toBeGreaterThan(section.indexOf('\n````\n'));
}
expect(text).toContain('"partial":N,"not_done":N');
expect(text).toContain('each Y response\'s evidence and each D response\'s dropped item');
expect(text).toContain('each Y\'d item with the user\'s free-text evidence and each D\'d item with "intentionally dropped"');
});
test('expands a body before the end marker in the skeleton', () => {
@@ -155,7 +177,7 @@ describe('workflow judge excerpts', () => {
const { skillPath, startMarker, endMarker } = ENG_REVIEW_EXCERPT;
const eng = readWorkflowExcerpt(skillPath, startMarker, endMarker);
const stages = ['## Review preparation', '## Retrospective learning', '## Confidence Calibration', '## Decision procedure',
'### 1. Establish current state', '## Scope Challenge', '### A. Assess the target',
'### Prepare an unanswered choice', '## Scope Challenge', '### A. Assess the target',
'### B. Resolve complexity selectors', '### C. Resolve findings', '## Review Sections',
'### 1. Architecture review', '### 2. Code quality review', '### 3. Test review', '### 4. Performance review']
.map(heading => eng.indexOf(heading));
@@ -164,11 +186,10 @@ describe('workflow judge excerpts', () => {
expect(eng.match(/^## Decision procedure$/gm)).toHaveLength(1);
const procedure = eng.slice(eng.indexOf('## Decision procedure'), eng.indexOf('## Scope Challenge'));
const headings = marked.lexer(procedure).filter(token => token.type === 'heading' && token.depth === 3);
expect(headings.map(token => token.text)).toEqual(['1. Establish current state', '2. Separate independent choices', '3. Compare one choice',
'4. Save the pending record', '5. Ask and wait', '6. Apply and refresh']);
expect(procedure).toContain("### 4. Save the pending record");
expect(procedure).toContain('### 5. Ask and wait');
expect(procedure).toContain("### 6. Apply and refresh");
expect(headings.map(token => token.text)).toEqual(['Prepare an unanswered choice', 'Send once and wait', 'Record the answer']);
expect(procedure).toContain("**Pending-record checkpoint.**");
expect(procedure).toContain('### Send once and wait');
expect(procedure).toContain("### Record the answer");
const outputs = ['### TODOS.md updates', '## Approval readiness', '## Required outputs', '## Implementation Tasks',
'### Unresolved decisions', '### Completion summary', '## Plan File Review Report',
'### Write to the report file', '## Review Log'].map(heading => eng.indexOf(heading));
@@ -268,7 +289,7 @@ console.log(JSON.stringify({calls, results}));
expectOutsideReviewControlFlow(eng, '**Construct the plan review prompt**');
expect(eng).toContain('Agreement between reviewers is evidence, not approval');
expect(eng).toContain('new or reopened choices still need their own answers');
const pendingDecision = eng.slice(eng.indexOf('### 5. Ask and wait'), eng.indexOf("### 6. Apply and refresh"));
const pendingDecision = eng.slice(eng.indexOf('### Send once and wait'), eng.indexOf("### Record the answer"));
expect(pendingDecision).toContain("**STOP until the actual answer arrives.**");
expect(pendingDecision.replace(/\s+/g, ' ')).toContain("Do not apply a remedy, make another call, start the next section or call ExitPlanMode while the choice awaits an answer");
expect(eng.replace(/\s+/g, ' ')).toContain("Use a scoped Edit to save this record and only the authorized working-plan amendments. Leave other choices unchanged");
@@ -322,8 +343,8 @@ console.log(JSON.stringify({calls, results}));
test('ship commits logical chunks without rewriting existing checkpoint commits', () => {
const text = readWorkflowExcerpt('ship/SKILL.md', '# Ship:', '## Important Rules');
const commit = text.slice(text.indexOf('## Step 15:'), text.indexOf('## Step 16:'));
expect(commit).toContain('Create small, logical commits');
expect(commit).toContain('If all changes are already committed, continue to Step 16');
expect(commit).toContain('Make bisectable commits');
expect(commit).toContain('if already committed, continue to Step 16');
expect(commit).toContain('Each commit must work independently');
expect(commit).not.toMatch(/rebase|reset|squash|fixup|WIP_TODO|gstack-context/);
expect(text).not.toContain('Step 15.0');