Merge remote-tracking branch 'origin/capy/audit-fix-wave' into capy/fixwave-baseline-repairs

This commit is contained in:
garrytan committed 2026-09-29 16:20:06 +00:00
commit a34457341b
3 files changed
+35 -3

No files matched your search

+1
View File
@@ -34,6 +34,7 @@ The table compares v1.91.6.0 with this branch before it merged v1.91.7.0, which
- The plan-count history PTY test waits for its startup marker instead of a fixed 8-second sleep.
### Fixed
- The `/plan-design-review` UI-scope gate eval recognizes a Design finding by the review's own issue-numbered options (`1A`, `1B`, …) as well as by UI vocabulary, so a real finding about hierarchy, navigation, state tables or confirmation patterns no longer goes uncounted until the 600-second cap. It timed out on v1.91.7.0 in one of two local runs and on this branch's CI; both fixed runs finished in about 420 seconds.
- `bun run test:ubicloud` no longer reports `pull failed` when a run leaves no flake ledger in `/tmp`: a retrieval glob that matches nothing is skipped with a note, and the retained shard logs still land in `.context/ubicloud/<timestamp>/free-test-logs/`.
### For contributors
+22
View File
@@ -134,6 +134,28 @@ mock.module(${JSON.stringify(path.join(ROOT, 'test/helpers/claude-pty-runner.ts'
expect(opts.isReviewAUQ(nativeFinding)).toBe(true);
expect(opts.isReviewAUQ(unnumberedFinding)).toBe(true);
expect(opts.isReviewAUQ(finding)).toBe(true);
// Titles and option-label prefixes projected from Sep 29 local captures of the
// real skill. The first four went unrecognized, so the run timed out after
// four answered findings; issue-numbered option sets identify each finding.
const labeled = (id, question, labels) => {
const call = fp(id, question, false);
call.nativeCall.questions[0].options = labels.map(label => ({ label, description: '' }));
return call;
};
for (const [question, labels] of [
['D7 — Issue 1: Codify the page hierarchy from approved Variant A in the plan?', ['1A) Full hierarchy with caps and overflo', '1B) Layout and read order only', '1C) Leave hierarchy out']],
['D8 — Issue 2: How does the dashboard fit the existing app shell and navigation?', ['2A) Reuse existing shell, Dashboard nav ', '2B) Standalone minimal header, no nav it', '2C) Leave the shell unspecified']],
['D9 — Issue 3: Add a user-visible interaction state table for every dashboard feature?', ['3A) Full state table with copy (recommen', '3B) Structure only, copy TBD', '3C) Leave states as listed']],
['D10 — Issue 4: Confirmation pattern and content for "Mark all as read"?', ['4A) Keep modal, specify copy and labels ', '4B) No modal; immediate action plus undo', '4C) Keep modal, leave copy to implemente']],
['Issue 1 — What does the user see first on a phone?', ['1A: Notifications first on sm, feed firs', '1B: Feed first everywhere', '1C: Bell badge + drawer on sm']],
]) expect(opts.isReviewAUQ(labeled('captured-issue', question, labels))).toBe(true);
for (const [question, labels] of [
['D11 — Update the approved mockups with these decisions?', ['A) Regenerate mockups (recommended)', 'B) Keep the current mockups']],
['D12 — Which follow-up should I record?', ['1A) Record a TODO', '2B) Skip it']],
['D13 — Which follow-up should I record?', ['1A) Record a TODO', '1A) Record it again']],
['D14 — Record this follow-up?', ['1A) Record a TODO']],
['D15 — Want outside design voices before the detailed review?', ['1A) Yes, run outside voices', '1B) No, proceed without']],
]) expect(opts.isReviewAUQ(labeled('not-a-finding', question, labels))).toBe(false);
const chosenFocus = mode.startsWith('native-') ? nativeFocus : mode === 'paraphrase' ? paraphrase : fp('focus', focus);
const pendingCall = {...chosenFocus.nativeCall, answered: false, unansweredQuestionIndices: [0]};
const pending = nativePlanCallFingerprint(pendingCall, 1000, true);
+12 -3
View File
@@ -33,14 +33,23 @@ const designFocusBoundary = (fp: AskUserQuestionFingerprint): boolean =>
});
// Require a choice about the supplied UI, not a workflow offer after focus.
// Both the question and an offered remedy must describe concrete UI behavior.
// The review labels each finding's options with its issue number and a letter
// ("3A", "3B"; review-sections.md), so a complete issue-labeled option set
// identifies a Design finding whatever vocabulary its title uses. Otherwise both
// the question and an offered remedy must describe concrete UI behavior.
const issueLabeled = (options: Array<{ label: string }>): boolean => {
const labels = options.map(({ label }) => /^\s*(\d+)([A-Z])(?=[):.\s]|$)/.exec(label));
return labels.length >= 2 && labels.every(Boolean)
&& new Set(labels.map(match => match![1])).size === 1
&& new Set(labels.map(match => match![2])).size === labels.length;
};
const uiChoice = /\b(?:layout|compos(?:e|ed|ition)|anchor|regions?|panels?|notifications?|activity|quick actions?|loading|skeletons?|empty|errors?|success|modals?|toasts?|buttons?|links?|copy|typography|fonts?|spacing|contrast|colors?|breakpoints?|responsive|keyboard|focus (?:order|trap|management)|aria|a11y|accessibility)\b/i;
const designReviewFinding = (fp: AskUserQuestionFingerprint): boolean =>
fp.nativeCall?.answered === true && !fp.nativeCall.failed && !designFocusBoundary(fp) && fp.nativeCall.questions.some(({ question, options }) => {
const title = question.split(/\r?\n/, 1)[0]!.trim().replace(/^D\d+(?:\.\d+)?\s*[—–:-]\s*/i, '');
const setup = /\b(?:outside (?:design )?voices|cross[ -]project learnings|review (?:target|scope|mode)|what should I (?:design[ -])?review|which (?:artifact|plan|file))\b/i;
return !setup.test(title) && uiChoice.test(title)
&& options.some(option => uiChoice.test(`${option.label} ${option.description}`));
return !setup.test(title) && (issueLabeled(options)
|| (uiChoice.test(title) && options.some(option => uiChoice.test(`${option.label} ${option.description}`))));
});
describeE2E('/plan-design-review with UI scope (gate)', () => {