diff --git a/.github/docker/Dockerfile.ci b/.github/docker/Dockerfile.ci index 4dc207e6b..0c4ca9a73 100644 --- a/.github/docker/Dockerfile.ci +++ b/.github/docker/Dockerfile.ci @@ -93,7 +93,13 @@ RUN curl --retry 5 --retry-delay 5 --retry-connrefused -fsSL https://bun.sh/inst # skillify HOME discovery on 2.1.237, guard/freeze hooks on 2.1.162). # Bump deliberately, via a PR that runs the PTY gate against the new TUI. # test/ci-image-cli-pin.test.ts fails the free suite if this pin is removed. -RUN npm i -g @anthropic-ai/claude-code@2.1.251 +# 2.1.284 (from 2.1.251, 2026-09-29): 2.1.251 logs +# [claude-code:unrecognized_model] for the eval model claude-fable-5-1; +# 2.1.284 recognizes it. Local canary: the gate PTY smoke subset parsed on +# both TUIs (7/8 pass on 2.1.284, 8/8 on 2.1.251; the one red was the model +# still in WebSearch at 300 s), and plan-design-review-plan-mode passed at +# 293 s on 2.1.284 where 2.1.251 timed out at 300 s. +RUN npm i -g @anthropic-ai/claude-code@2.1.284 # Playwright system deps (Chromium) — needed for browse E2E tests RUN npx playwright install-deps chromium diff --git a/deslop-shared-libs/SKILL.md b/deslop-shared-libs/SKILL.md index ef62064fd..7177ef891 100644 --- a/deslop-shared-libs/SKILL.md +++ b/deslop-shared-libs/SKILL.md @@ -115,6 +115,8 @@ the repo, traverse submodule worktrees, or execute filters. Note excluded symlin submodule, ignored, unavailable or unreadable source. Handle deletions explicitly. Do not call an absent or unreadable overlay clean. Current raw content may differ even when a clean filter would produce the same Git tree. +Sessions have a bounded number of turns. Read related files together: parallel +host reads or one read-only command per step, not one file per turn. ## Start with recent work diff --git a/deslop-shared-libs/SKILL.md.tmpl b/deslop-shared-libs/SKILL.md.tmpl index 1d8a3329b..15c1f3b65 100644 --- a/deslop-shared-libs/SKILL.md.tmpl +++ b/deslop-shared-libs/SKILL.md.tmpl @@ -109,6 +109,8 @@ the repo, traverse submodule worktrees, or execute filters. Note excluded symlin submodule, ignored, unavailable or unreadable source. Handle deletions explicitly. Do not call an absent or unreadable overlay clean. Current raw content may differ even when a clean filter would produce the same Git tree. +Sessions have a bounded number of turns. Read related files together: parallel +host reads or one read-only command per step, not one file per turn. ## Start with recent work diff --git a/plan-design-review/SKILL.md b/plan-design-review/SKILL.md index 48574b18c..cf0991f32 100644 --- a/plan-design-review/SKILL.md +++ b/plan-design-review/SKILL.md @@ -767,6 +767,8 @@ review design — real visuals, not text descriptions." The ONLY time you skip mockups is when: - `DESIGN_NOT_AVAILABLE` was printed (designer binary not found) +- The first `$D` generation command fails before producing an image (for + example `No OpenAI API key found`): treat it exactly as `DESIGN_NOT_AVAILABLE` - The plan has zero UI scope (pure backend/API/infrastructure) If the user explicitly says "skip mockups" or "text only", respect that. Otherwise, generate. @@ -928,7 +930,7 @@ Note which direction was approved. This becomes the visual reference for all sub **Multiple variants/screens:** If the user asked for multiple variants (e.g., "5 versions of the homepage"), generate ALL as separate variant sets with their own comparison boards. Each screen/variant set gets its own subdirectory under `designs/`. Complete all mockup generation and user selection before starting review passes. -**If `DESIGN_NOT_AVAILABLE`:** Tell the user: "The gstack designer isn't set up yet. Run `$D setup` to enable visual mockups. Proceeding with text-only review, but you're missing the best part." Then proceed to review passes with text-based review. +**If `DESIGN_NOT_AVAILABLE`:** Tell the user: "The gstack designer isn't set up yet. Run `$D setup` to enable visual mockups. Proceeding with text-only review, but you're missing the best part." Then proceed to review passes with text-based review. Do not substitute hand-built HTML/CSS wireframes, screenshots or a comparison board of your own: they delay the first review question by minutes and are not designer output. ## Design Outside Voices (independent) diff --git a/plan-design-review/SKILL.md.tmpl b/plan-design-review/SKILL.md.tmpl index db88a19fb..5ce6b5341 100644 --- a/plan-design-review/SKILL.md.tmpl +++ b/plan-design-review/SKILL.md.tmpl @@ -205,6 +205,8 @@ review design — real visuals, not text descriptions." The ONLY time you skip mockups is when: - `DESIGN_NOT_AVAILABLE` was printed (designer binary not found) +- The first `$D` generation command fails before producing an image (for + example `No OpenAI API key found`): treat it exactly as `DESIGN_NOT_AVAILABLE` - The plan has zero UI scope (pure backend/API/infrastructure) If the user explicitly says "skip mockups" or "text only", respect that. Otherwise, generate. @@ -264,7 +266,7 @@ Note which direction was approved. This becomes the visual reference for all sub **Multiple variants/screens:** If the user asked for multiple variants (e.g., "5 versions of the homepage"), generate ALL as separate variant sets with their own comparison boards. Each screen/variant set gets its own subdirectory under `designs/`. Complete all mockup generation and user selection before starting review passes. -**If `DESIGN_NOT_AVAILABLE`:** Tell the user: "The gstack designer isn't set up yet. Run `$D setup` to enable visual mockups. Proceeding with text-only review, but you're missing the best part." Then proceed to review passes with text-based review. +**If `DESIGN_NOT_AVAILABLE`:** Tell the user: "The gstack designer isn't set up yet. Run `$D setup` to enable visual mockups. Proceeding with text-only review, but you're missing the best part." Then proceed to review passes with text-based review. Do not substitute hand-built HTML/CSS wireframes, screenshots or a comparison board of your own: they delay the first review question by minutes and are not designer output. {{DESIGN_OUTSIDE_VOICES}} diff --git a/test/arm-benchmark-selftest.test.ts b/test/arm-benchmark-selftest.test.ts index e622bf645..347d1e8f5 100644 --- a/test/arm-benchmark-selftest.test.ts +++ b/test/arm-benchmark-selftest.test.ts @@ -13,7 +13,7 @@ import { } from './helpers/arm-benchmark-harness'; import { armJudge, buildArmJudgePrompt, parseArmJudgeResponse, - ARM_JUDGE_ATTEMPTS, callJudge, + callJudge, } from './helpers/llm-judge'; import * as fs from 'fs'; import * as path from 'path'; @@ -182,28 +182,25 @@ describe('arm benchmark selftest (free, no API)', () => { expect(score.construct).toBe('none'); }); - test('armJudge: bounded retry-on-malformed — recovers once, then gives up', async () => { - // Malformed first, valid second: recovers within the 2-attempt bound. + test('armJudge: a malformed verdict is a failed sample, never re-asked', async () => { let calls = 0; - const flaky = (async () => { + const malformedFirst = (async () => { calls++; return calls === 1 ? { over_engineering: 9, construct: 'garbage' } : { over_engineering: 2, construct: 'repository layer in app.js', reasoning: 'ok' }; }) as unknown as typeof callJudge; - const recovered = await armJudge('ticket', 'diff --git a/x b/x\n+1\n', { call: flaky }); - expect(recovered.over_engineering).toBe(2); - expect(calls).toBe(ARM_JUDGE_ATTEMPTS); + await expect(armJudge('ticket', 'diff --git a/x b/x\n+1\n', { call: malformedFirst })) + .rejects.toThrow(/malformed verdict \(never resampled\)/); + expect(calls).toBe(1); - // Always malformed: throws after exactly ARM_JUDGE_ATTEMPTS attempts. - let badCalls = 0; - const alwaysBad = (async () => { - badCalls++; - return { nonsense: true }; + let goodCalls = 0; + const wellFormed = (async () => { + goodCalls++; + return { over_engineering: 2, construct: 'repository layer in app.js', reasoning: 'ok' }; }) as unknown as typeof callJudge; - await expect(armJudge('ticket', 'diff --git a/x b/x\n+1\n', { call: alwaysBad })) - .rejects.toThrow(/no well-formed verdict after 2 attempts/); - expect(badCalls).toBe(ARM_JUDGE_ATTEMPTS); + expect((await armJudge('ticket', 'diff --git a/x b/x\n+1\n', { call: wellFormed })).over_engineering).toBe(2); + expect(goodCalls).toBe(1); }); }); diff --git a/test/ceo-mode-option.test.ts b/test/ceo-mode-option.test.ts index edcc8f8ad..cacfec1c9 100644 --- a/test/ceo-mode-option.test.ts +++ b/test/ceo-mode-option.test.ts @@ -10,6 +10,7 @@ import captured_ceo_hold_commitment_ar from './fixtures/ceo-hold-commitment-ar.j import captured_ceo_hold_posture_ag from './fixtures/ceo-hold-posture-ag.json'; import retainedPreservationCaptures_ceo_hold_posture_ag from './fixtures/ceo-hold-preservation-f359.json'; import captured_ceo_mode_colon_at from './fixtures/ceo-mode-colon-at.json'; +import scrolledReview from './fixtures/ceo-mode-scrolled-review-36606688266.json'; import fs_ceo_mode_full_ad from 'node:fs'; import os_ceo_mode_full_ad from 'node:os'; import path_ceo_mode_full_ad from 'node:path'; @@ -1840,3 +1841,55 @@ test('AD v2 prerequisite requires the active native packet identity',()=>{ for(const delta of [{answered:true},{failed:true},{sessionId:''},{toolUseId:''}]){const call={...pending(),...delta};const x=frame(call,2);expect(planCountPrerequisitePick(x.routing,x.active)).toBeNull();} }); }); + +describe('mode submission when the review panel scrolls past the viewport', () => { + // Run 36606688266 bundled routing, learnings and the mode choice into one + // native call. Its review panel was taller than the terminal, so the tab bar + // scrolled away and the harness never submitted HOLD SCOPE. + const scrolledTranscript = scrolledReview.transcript as unknown as PlanCountTranscript; + const scrolledCall = scrolledTranscript.calls[0] as NativePlanQuestionCall; + const scrolledSubmit = (screen: string, screenText: string, mode: 'HOLD SCOPE' | 'SCOPE EXPANSION' = 'HOLD SCOPE', + selected: NativePlanQuestionCall = scrolledCall, native: PlanCountTranscript = scrolledTranscript) => + ceoModeSubmissionInput(screen, selected, mode, native, new Set(), screenText); + + test('the captured viewport has no tab bar and ends at the focused Submit prompt', () => { + expect(scrolledReview.screen).not.toMatch(/←[^\r\n]+✔\s*Submit\s*→/); + expect(scrolledReview.screen.trimEnd()).toMatch(/❯ 1\. Submit answers\s+2\. Cancel$/); + expect(scrolledCall.questions.map(q => q.header)).toEqual(['Routing', 'Learnings', 'Review mode']); + }); + + test('the complete scrolled review submits the selected mode once', () => { + expect(scrolledSubmit(scrolledReview.screen, scrolledReview.screenText)).toBe('\r'); + const seen = new Set(); + expect(ceoModeSubmissionInput(scrolledReview.screen, scrolledCall, 'HOLD SCOPE', scrolledTranscript, seen, scrolledReview.screenText)).toBe('\r'); + expect(ceoModeSubmissionInput(scrolledReview.screen, scrolledCall, 'HOLD SCOPE', scrolledTranscript, seen, scrolledReview.screenText)).toBeNull(); + }); + + test('without the accumulated screen text a barless viewport cannot submit', () => { + expect(scrolledSubmit(scrolledReview.screen, '')).toBeNull(); + }); + + test('a review showing another mode is not an acknowledgement of the target mode', () => { + expect(scrolledSubmit(scrolledReview.screen, scrolledReview.screenText, 'SCOPE EXPANSION')).toBeNull(); + }); + + for (const [name, change] of [ + ['an answer no option offers', (text: string) => text.replace(/→ Enable cross-project \(recommended\)(?![\s\S]*→ Enable cross-project)/, '→ Upload learnings')], + ['an altered question', (text: string) => text.replace(/D2 — Let gstack(?![\s\S]*D2 — Let gstack)/, 'D2 — Never let gstack')], + ['a quoted review', (text: string) => text.replace(/Review your answers(?![\s\S]*Review your answers)/, 'Quoted example:\nReview your answers')], + ['output after the prompt', (text: string) => `${text}\nMore text`], + ] as const) test(`the scrolled route rejects ${name}`, () => { + expect(scrolledSubmit(scrolledReview.screen, change(scrolledReview.screenText))).toBeNull(); + }); + + test('the viewport must still end at the focused Submit prompt', () => { + expect(scrolledSubmit(scrolledReview.screen.replace('❯ 1. Submit answers', ' 1. Submit answers\n❯ 2. Cancel'), scrolledReview.screenText)).toBeNull(); + }); + + test('an answered or changed native call cannot be submitted again', () => { + expect(scrolledSubmit(scrolledReview.screen, scrolledReview.screenText, 'HOLD SCOPE', { ...scrolledCall, answered: true })).toBeNull(); + const other = structuredClone(scrolledCall); + other.questions[1]!.question += ' (changed)'; + expect(scrolledSubmit(scrolledReview.screen, scrolledReview.screenText, 'HOLD SCOPE', other)).toBeNull(); + }); +}); diff --git a/test/eng-seeded-coverage.test.ts b/test/eng-seeded-coverage.test.ts index 66a2d07cc..59df5a208 100644 --- a/test/eng-seeded-coverage.test.ts +++ b/test/eng-seeded-coverage.test.ts @@ -1,7 +1,12 @@ import { describe, expect, test } from 'bun:test'; import type { NativePlanQuestionCall } from './helpers/plan-count-transcript'; -import { isEngBatchingIssueAUQ } from './helpers/eng-seeded-coverage'; -import { nativePlanCallFingerprint } from './helpers/claude-pty-runner'; +import * as fs from 'node:fs'; +import * as os from 'node:os'; +import * as path from 'node:path'; +import { createEngBatchingIssueCounter, isEngBatchingIssueAUQ } from './helpers/eng-seeded-coverage'; +import { engSetupAUQ, hasCompletePlanReport, nativePlanCallFingerprint } from './helpers/claude-pty-runner'; +import batchingCapture from './fixtures/eng-batching-unsourced-brief-36606688266.json'; +import bulletTargetCapture from './fixtures/eng-batching-bullet-target-rerun.json'; function question(call: NativePlanQuestionCall, text: string) { const answer = call.answers![call.questions[0]!.question]!; @@ -78,3 +83,112 @@ describe('batching caller counts completed issue decisions across setup boundari expect(check(quoted)).toBe(true); }); }); + +describe('batching replay of run 36606688266 (unsourced native briefs)', () => { + // Run 36606688266 asked one native question per finding (D1-D9 bound to + // ledger records R1-R9, D10 a TODO follow-up) but cited no PLAN.md line in the + // native brief, so the old detector counted zero review decisions. + const FLOOR = 3; + const calls = batchingCapture.calls as unknown as NativePlanQuestionCall[]; + + function count(plan: string, edit: (calls: NativePlanQuestionCall[]) => void = () => {}) { + const copy = structuredClone(calls); + edit(copy); + const counter = createEngBatchingIssueCounter(() => plan, engSetupAUQ); + const counted = copy.filter((call, index) => counter.isReviewAUQ(nativePlanCallFingerprint(call, 0, true), copy.slice(0, index))); + return { counted: counted.length, issues: counter.trace.map(entry => entry.issue) }; + } + + test('the recorded failing verdict is the detector, not the review', () => { + expect(batchingCapture.recordedOutcome).toEqual({ outcome: 'completion_summary', step0Count: 10, reviewCount: 0 }); + expect(calls.every(call => call.answered && call.questions.length === 1)).toBe(true); + }); + + test('each ledger-bound native decision counts once without a native source citation', () => { + const { counted, issues } = count(batchingCapture.plan); + expect(issues).toEqual(['R1', 'R2', 'R3', 'R4', 'R5', 'R6', 'R7', 'R8', 'R9'].map(id => `record:${id}`)); + expect(counted).toBeGreaterThanOrEqual(FLOOR); + }); + + test('a re-asked decision cannot inflate the count', () => { + const { counted } = count(batchingCapture.plan, all => { + const again = structuredClone(all[0]!); + again.toolUseId += '-again'; + all.splice(1, 0, again); + }); + expect(counted).toBe(9); + }); + + const target = 'Review target (fixed): `PLAN.md`'; + for (const [name, plan] of [ + ['a foreign target', batchingCapture.plan.replace(target, 'Review target (fixed): `OTHER.md`')], + ['a mixed target', batchingCapture.plan.replace(target, 'Review target (fixed): `OTHER.md` and `PLAN.md`')], + ['two target declarations', batchingCapture.plan.replace(target, `${target}\nReview target (fixed): \`PLAN.md\``)], + ['no target declaration', batchingCapture.plan.replace(target, 'Report scope: the fixture repo')], + ['a report title for another plan', batchingCapture.plan.replace('# Engineering review: Add background job retry framework', '# Engineering review: Replace all customer data')], + ['an archived report title', batchingCapture.plan.replace('# Engineering review:', '# Archived engineering review:')], + ['a copied H1 naming another plan', batchingCapture.plan.replace('# Plan: Add background job retry framework', '# Plan: Replace all customer data')], + ] as const) test(`the unsourced route rejects ${name}`, () => { + expect(count(plan).counted).toBe(0); + }); + + test('the unsourced route rejects a native brief naming another plan or file', () => { + const rename = (from: string, to: string) => (all: NativePlanQuestionCall[]) => { + for (const call of all) call.questions[0]!.question = call.questions[0]!.question.replace(from, to); + }; + expect(count(batchingCapture.plan, rename('plan "Add background job retry framework"', 'plan "Replace all customer data"')).counted).toBe(0); + expect(count(batchingCapture.plan, rename('plan "Add background job retry framework"', 'plan "Add background job retry framework", OTHER.md')).counted).toBe(0); + expect(count(batchingCapture.plan, rename('plan "Add background job retry framework"', 'the plan')).counted).toBe(0); + }); + + test('a saved record whose brief title differs from the native question does not bind it', () => { + const plan = batchingCapture.plan.replace(/^Question D1:\n.*$/m, 'Question D1:\nD1 — Some other decision?'); + expect(count(plan).issues).not.toContain('record:R1'); + }); + + test('the completed report is the early outcome point; a partial report is not', () => { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'eng-batching-report-')); + try { + const report = path.join(dir, 'report.md'); + fs.writeFileSync(report, batchingCapture.plan); + expect(hasCompletePlanReport(report, 0, Date.now() + 1_000)).toBe(true); + fs.writeFileSync(report, batchingCapture.plan.slice(0, batchingCapture.plan.indexOf('## Completion summary'))); + expect(hasCompletePlanReport(report, 0, Date.now() + 1_000)).toBe(false); + fs.writeFileSync(report, batchingCapture.plan.replace('## GSTACK REVIEW REPORT', '```\n## GSTACK REVIEW REPORT') + '\n```\n'); + expect(hasCompletePlanReport(report, 0, Date.now() + 1_000)).toBe(false); + } finally { + fs.rmSync(dir, { recursive: true, force: true }); + } + }); +}); + +describe('batching replay of a 2.1.284 rerun (bullet target, unnamed plan)', () => { + // Eleven separate native questions; the briefs name no plan and the report + // declares '- **Review target (fixed):** `/abs/PLAN.md`' under '# Eng Review — PLAN.md: '. + const calls = bulletTargetCapture.calls as unknown as NativePlanQuestionCall[]; + const count = (plan: string) => { + const counter = createEngBatchingIssueCounter(() => plan, engSetupAUQ); + calls.forEach((call, index) => counter.isReviewAUQ(nativePlanCallFingerprint(call, 0, true), calls.slice(0, index))); + return counter.trace.map(entry => entry.issue); + }; + + test('the recorded verdict counted none of the separate decisions', () => { + expect(bulletTargetCapture.recordedOutcome).toMatchObject({ reviewCount: 0 }); + expect(calls.length).toBe(11); + }); + + test('ledger-bound decisions count once each through the report target field', () => { + expect(count(bulletTargetCapture.plan).length).toBe(9); + }); + + for (const [name, change] of [ + ['a foreign target file', (plan: string) => plan.replace(/(Review target \(fixed\):\*\* `[^`]*\/)PLAN\.md`/, '$1OTHER.md`')], + ['a second target declaration', (plan: string) => plan.replace('- **Review target (fixed):**', '- **Review target (fixed):** `OTHER.md`\n- **Review target (fixed):**')], + ['no target declaration', (plan: string) => plan.replace('- **Review target (fixed):**', '- **Report scope:**')], + ['an archived report title', (plan: string) => plan.replace('# Eng Review —', '# Archived Eng Review —')], + ] as const) test(`the bullet target route rejects ${name}`, () => { + const plan = change(bulletTargetCapture.plan); + expect(plan).not.toBe(bulletTargetCapture.plan); + expect(count(plan)).toEqual([]); + }); +}); diff --git a/test/fixtures/ceo-mode-scrolled-review-36606688266.json b/test/fixtures/ceo-mode-scrolled-review-36606688266.json new file mode 100644 index 000000000..e908b4644 --- /dev/null +++ b/test/fixtures/ceo-mode-scrolled-review-36606688266.json @@ -0,0 +1,72 @@ +{ + "source": "run 36606688266 plan-ceo-mode-routing HOLD SCOPE: final viewport, accumulated screen text from the last tab frame, and the pending native call", + "screen": " \u2502 ELI10: gstack works best when your project's CLAUDE.md includes skill routing rules, so requests like \"review this\n \u2502 diff\" route to the right skill automatically. This is a plain text section appended to CLAUDE.md.\n \u2502 Stakes if we pick wrong: without it you invoke skills by name manually; with it, plain requests auto-route. Either\n \u2502 is reversible.\n \u2502 Recommendation: A because auto-routing removes a step from every future session and costs one commit.\n \u2502 Note: options differ in kind, not coverage \u2014 no completeness score.\n \u2502 Net: convenience now vs. one extra committed section in CLAUDE.md. Plan mode blocks file edits, so if you pick A\n \u2502 the append + commit happens after this review exits plan mode.\n \u2192 Add routing rules (recommended)\n \u2502 \u25cf D2 \u2014 Let gstack search learnings from your other projects on this machine?\n \u2502 Project/branch/task: gstack-plan-count-FwyQuk on main; one-time gstack setup prompt.\n \u2502 ELI10: gstack saves small lessons per project (\"this test runner needs flag X\"). Cross-project mode also searches\n \u2502 lessons from your other local projects when reviewing this one. Everything stays on this machine.\n \u2502 Stakes if we pick wrong: too narrow and you miss patterns you already learned elsewhere; too broad and a client\n \u2502 codebase could surface a lesson from another client's repo in a review.\n \u2502 Recommendation: A because this is a solo-style environment and the data never leaves the machine.\n \u2502 Note: options differ in kind, not coverage \u2014 no completeness score.\n \u2502 Net: more recall vs. strict per-project isolation.\n \u2192 Enable cross-project (recommended)\n \u2502 \u25cf D3 \u2014 R2: Which review mode for the saved-views plan?\n \u2502 Project/branch/task: gstack-plan-count-FwyQuk on main; reviewing PLAN.md \"Add saved project views\".\n \u2502 ELI10: The mode sets my posture for the rest of the review. Expansion pushes for the biggest version, Hold Scope\n \u2502 stress-tests exactly what you wrote, Reduction strips to the smallest shippable core, and Selective holds your\n \u2502 scope while offering a few add-ons one at a time for you to accept or decline.\n \u2502 Stakes if we pick wrong: too ambitious and a small feature balloons; too strict and we ship personal-only views\n \u2502 when the goal (\"team members repeatedly recreate filters\") may really be a shared-view problem, forcing a second\n \u2502 migration later.\n \u2502 Recommendation: SELECTIVE EXPANSION because the plan is an added capability of ~8\u201310 files, but its member-only\n \u2502 scoping is the one fact that could be wrong: every incumbent ships shared views too, and the table shape decides\n \u2502 whether adding them later is a column or a rewrite. Selective lets you rule on that once without committing to a\n \u2502 bigger build.\n \u2502 Note: options differ in kind, not coverage \u2014 no completeness score.\n \u2502 Net: how much of the review is spent challenging scope vs. hardening the scope you already chose.\n \u2192 HOLD SCOPE\n\nReady to submit your answers?\n\n\u276f 1. Submit answers\n 2. Cancel\n", + "screenText": "omething.\n\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n 4. Chat about this\n\nEnter to select \u00b7 Tab/Arrow keys to navigate \u00b7 Esc to cancel\n\n\n\n \u2612 Learnings \u2610 Review mode \n3R2:Which review mode for the saved-views plan?\nreviewing PLAN.md \"Add saved project views\".\nThe mode sts y posture for the rest of e review. Expansion pushes forthe biggest version, Hld Scop \nstress-testsexactly what youwt, Reduction strpso the smallst shippable cre, and Selective holds your scope \nwhil ofering a few add-onsone at time for you o accept or dcline.\nStakes if we pick wrong:too ambitious and asmall featureballoons; too strict and we ship personal-only views when \nthe goal (\"teammembers repeatedly recreate filters\") ay really be shared-viw problm, forcing a second migration \nlar.\nRcomendation: SELECTIVE EXPANSION becaue the plan is an added capability of ~8\u201310 files, but its member-only \n\u2502scoping is the one fact that could be wrong: every incumbent ships shared views too, and the table shape decides \n\u2502whether adding them later is a column or a rewrite. Selective lets you rule on that once without committing to a \n\u2502bigger build.\n\u2502Note: options differ in kind, not coverage \u2014 no completeness score.\n\u2502Net: how much of the review is spent challenging scope vs. hardening the scope you already chose.\n\n\u276f1.SELECTIVE EXPANSION (recommended)\n\u2705 Keep your fou approach bullets as the bselineand hardens hemwith fullrigor\ufffd\u2705 Offers each expansion \n (shared views, default view, cleanup) as a separate add/defer/skip call\ufffd\u274c A few more decision questions than Hold \n Scope before the deep review starts\n2.HOLDSCOPE\n\u2705 Maximum rigor on exactly what is written: error paths, edge cases, tests, observability\ufffd\u2705 Fastest path to an \n implementation-ready plan wiho scopquetions\ufffd\u274c Shared views and table-shape futureproofing get flagged, not \noffered; possible second migration later\n\n3.SCOPEEXPANSION\n\n\u2705Designstheplatonicsaved-viewsfeature:personal+shared,defaults,sharelinks,cleanup\ufffd\u2705Bestlong-term\n\narchitectureupfront;nofollow-upmigrations\ufffd\u274cTurnsa~10-filefeatureintoamulti-surfacebuildbeforethe\n\ntwo-weekpilotprovesreuse\n\n4.SCOPEREDUCTION\n\n\u2705Findsthesmallestcorethatteststhepilothypothesis(maybecreate/list/applyonly)\ufffd\u2705Lowestriskand\n\nfastesttothetwo-weekreusemeasurement\ufffd\u274cUpdate/deleteandpickerpolishgetdeferred;pilotmaymeasurea\n\nclunkyversionofthefeature\n\n5.Typesomething.\n\n\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n\n6.Chataboutthis\n\n\n\nEntertoselect\u00b7Tab/Arrowkeystonavigate\u00b7Esctocancel\n\n\n\nReview your answers\n \u2502 \u25cf D1 \u2014 Add gstack skill routing rules to this project'sCLAUDE.md?\n \u2502 Project/branch/task: gstack-plan-count-FwyQuk on main; one-time gstack setup prompt.\n \u2502 ELI10: gstack works best when your project's CLAUDE.md includes skill routing rules, so requests like \"review this\n \u2502 diff\" route to the right skill automatically. This is a plain text section appended to CLAUDE.md.\n \u2502 Stakes if we pick wrong: without it you invoke skills by name manually;withit,plainrequestsauto-route.Either\n \u2502 is reversible.\n \u2502 Recommendation: A because auto-routing removes a step from every future session and costs one commit.\n \u2502 Note:optionsdifferinkind,notcoverage\u2014nocompletenessscore.\n \u2502 Net: convenience now vs. one extra committed section in CLAUDE.md. Plan mode blocks file edits, so if you pickA\n \u2502 the append + commit happens after this review exits plan mode.\n \u2192 Add routing rules (recommended)\n \u2502 \u25cf D2 \u2014 Let gstacksearchlearningsfromyourotherprojectsonthismachine?\n \u2502 Project/branch/task: gstack-plan-count-FwyQuk on main; one-time gstacksetupprompt.\n \u2502 ELI10: gstack saves small lessons per project (\"this test runner needs flag X\"). Cross-projectmodealsosearches\n\u2502lessonsfromyourotherlocalprojectswhenreviewingthisone.Everythingstaysonthismachine.\n \u2502 Stakes if we pick wrong: too narrowandyoumisspatternsyoualreadylearnedelsewhere;toobroadandaclient\n\u2502codebase could surface a lesson from another client's repo in a review.\n\u2502Recommendation: A because this is a solo-style environment and the data never leaves the machine.\n\u2502Note: options differ in kind, not coverage\u2014nocompletenessscore.\n\u2502 Net:more recallvs.strictper-projectisolation.\n\u2192 Enable cross-project (recommended)\n\u2502\u25cfD3 \u2014 R2: Which review mode for the saved-views plan?\n\u2502Project/branch/task: gstack-plan-count-FwyQukonmain;reviewingPLAN.md\"Addsavedprojectviews\".\n\u2502 ELI10: The modesetsmyposturefortherestofthereview.Expansionpushesforthebiggestversion,HoldScope\n\u2502stress-tests exactly what you wrote, Reduction strips to the smallest shippable core, and Selective holds your\n\u2502scope while offering a few add-ons one at a time for you to accept or decline.\n\u2502Stakes if we pick wrong: tooambitiousandasmallfeatureballoons;toostrictandweshippersonal-onlyviews\n\u2502 when the goal (\"teammembersrepeatedlyrecreatefilters\")mayreallybeashared-viewproblem,forcingasecond\n\u2502migration later.\n\u2502Recommendation: SELECTIVE EXPANSION because the plan is an added capability of ~8\u201310 files, but its member-only\n\u2502scoping is the one fact that couldbewrong:everyincumbentshipssharedviewstoo,andthetableshapedecides\n\u2502 whether adding themlaterisacolumnorarewrite.Selectiveletsyouruleonthatoncewithoutcommittingtoa\n \u2502 bigger build.\n\u2502 Note: options differinkind,notcoverage\u2014nocompletenessscore.\n\u2502Net:howmuchofthereviewisspentchallengingscopevs.hardeningthescopeyoualreadychose.\n \u2192 HOLD SCOPE\n\n\n\nReadytosubmityouranswers?\n\n\n\n\u276f1.Submitanswers\n\n2.Cancel\n\n", + "transcript": { + "status": "ready", + "calls": [ + { + "sessionId": "4ea226bf-5a31-4342-8141-c76094320f94", + "toolUseId": "toolu_01Ee4tS4SoMXqLyS14ayssEy", + "questions": [ + { + "question": "D1 \u2014 Add gstack skill routing rules to this project's CLAUDE.md?\nProject/branch/task: gstack-plan-count-FwyQuk on main; one-time gstack setup prompt.\nELI10: gstack works best when your project's CLAUDE.md includes skill routing rules, so requests like \"review this diff\" route to the right skill automatically. This is a plain text section appended to CLAUDE.md.\nStakes if we pick wrong: without it you invoke skills by name manually; with it, plain requests auto-route. Either is reversible.\nRecommendation: A because auto-routing removes a step from every future session and costs one commit.\nNote: options differ in kind, not coverage \u2014 no completeness score.\nNet: convenience now vs. one extra committed section in CLAUDE.md. Plan mode blocks file edits, so if you pick A the append + commit happens after this review exits plan mode.", + "header": "Routing", + "multiSelect": false, + "options": [ + { + "label": "Add routing rules (recommended)", + "description": "\u2705 Plain requests (\"ship this\", \"find the bug\") route to the right skill without naming it\n\u2705 One short section, committed once; teammates get the same routing\n\u274c Adds a gstack-specific block to CLAUDE.md; applied after plan mode exits, not now" + }, + { + "label": "No thanks, manual invocation", + "description": "\u2705 CLAUDE.md stays exactly as it is; no tool-specific text in the repo\n\u2705 You keep full control over when a skill runs\n\u274c Every skill needs its /name typed; re-enable later via gstack-config routing_declined false" + } + ] + }, + { + "question": "D2 \u2014 Let gstack search learnings from your other projects on this machine?\nProject/branch/task: gstack-plan-count-FwyQuk on main; one-time gstack setup prompt.\nELI10: gstack saves small lessons per project (\"this test runner needs flag X\"). Cross-project mode also searches lessons from your other local projects when reviewing this one. Everything stays on this machine.\nStakes if we pick wrong: too narrow and you miss patterns you already learned elsewhere; too broad and a client codebase could surface a lesson from another client's repo in a review.\nRecommendation: A because this is a solo-style environment and the data never leaves the machine.\nNote: options differ in kind, not coverage \u2014 no completeness score.\nNet: more recall vs. strict per-project isolation.", + "header": "Learnings", + "multiSelect": false, + "options": [ + { + "label": "Enable cross-project (recommended)", + "description": "\u2705 Lessons from other local projects surface here when they match a finding\n\u2705 Local only; nothing leaves the machine, and it is a one-line config flip to undo\n\u274c Not ideal if you work on multiple client codebases where mixing context is a concern" + }, + { + "label": "Keep project-scoped only", + "description": "\u2705 Strict isolation: reviews here only see this project's lessons\n\u2705 Safe default for multi-client work or shared machines\n\u274c Patterns you already learned on another repo will not be suggested here" + } + ] + }, + { + "question": "D3 \u2014 R2: Which review mode for the saved-views plan?\nProject/branch/task: gstack-plan-count-FwyQuk on main; reviewing PLAN.md \"Add saved project views\".\nELI10: The mode sets my posture for the rest of the review. Expansion pushes for the biggest version, Hold Scope stress-tests exactly what you wrote, Reduction strips to the smallest shippable core, and Selective holds your scope while offering a few add-ons one at a time for you to accept or decline.\nStakes if we pick wrong: too ambitious and a small feature balloons; too strict and we ship personal-only views when the goal (\"team members repeatedly recreate filters\") may really be a shared-view problem, forcing a second migration later.\nRecommendation: SELECTIVE EXPANSION because the plan is an added capability of ~8\u201310 files, but its member-only scoping is the one fact that could be wrong: every incumbent ships shared views too, and the table shape decides whether adding them later is a column or a rewrite. Selective lets you rule on that once without committing to a bigger build.\nNote: options differ in kind, not coverage \u2014 no completeness score.\nNet: how much of the review is spent challenging scope vs. hardening the scope you already chose.", + "header": "Review mode", + "multiSelect": false, + "options": [ + { + "label": "SELECTIVE EXPANSION (recommended)", + "description": "\u2705 Keeps your four approach bullets as the baseline and hardens them with full rigor\n\u2705 Offers each expansion (shared views, default view, cleanup) as a separate add/defer/skip call\n\u274c A few more decision questions than Hold Scope before the deep review starts" + }, + { + "label": "HOLD SCOPE", + "description": "\u2705 Maximum rigor on exactly what is written: error paths, edge cases, tests, observability\n\u2705 Fastest path to an implementation-ready plan with no scope questions\n\u274c Shared views and table-shape futureproofing get flagged, not offered; possible second migration later" + }, + { + "label": "SCOPE EXPANSION", + "description": "\u2705 Designs the platonic saved-views feature: personal + shared, defaults, share links, cleanup\n\u2705 Best long-term architecture up front; no follow-up migrations\n\u274c Turns a ~10-file feature into a multi-surface build before the two-week pilot proves reuse" + }, + { + "label": "SCOPE REDUCTION", + "description": "\u2705 Finds the smallest core that tests the pilot hypothesis (maybe create/list/apply only)\n\u2705 Lowest risk and fastest to the two-week reuse measurement\n\u274c Update/delete and picker polish get deferred; pilot may measure a clunky version of the feature" + } + ] + } + ], + "answered": false, + "failed": false + } + ], + "assistantMessages": [] + } +} \ No newline at end of file diff --git a/test/fixtures/eng-batching-bullet-target-rerun.json b/test/fixtures/eng-batching-bullet-target-rerun.json new file mode 100644 index 000000000..1a1e318a6 --- /dev/null +++ b/test/fixtures/eng-batching-bullet-target-rerun.json @@ -0,0 +1,359 @@ +{ + "source": "local targeted rerun smoke-2.1.284-1790711269 (Claude Code 2.1.284) of plan-eng-multi-finding-batching: observation.json transcript.calls and the saved report replayed from its Write/Edit inputs", + "recordedOutcome": { + "outcome": "collection_complete", + "step0Count": 11, + "reviewCount": 0 + }, + "calls": [ + { + "sessionId": "0117efe3-a002-42c1-aa59-d45032643e13", + "toolUseId": "toolu_01XnWRrh4F44QdznytAGmriy", + "questions": [ + { + "question": "D1 \u2014 Reuse the job library's retry hooks or roll a custom scheduler?\nProject/branch/task: main branch, retry-framework plan; adding retries to 5 background workers.\nELI10: The job library you already use has retry hooks built in, and your plan says your custom version would be \"the same shape.\" Building your own copy inside each worker means five hand-written schedulers to keep correct, versus configuring one curve the library already knows how to run. The plan's reason for going custom is \"full control over the curve,\" and most retry hook APIs give you that through a backoff callback.\nStakes if we pick wrong: five bespoke schedulers drift apart, each grows its own bugs (no jitter, no cap, retry storms), and nobody at 3am knows which curve a given worker actually runs.\nRecommendation: A because the plan admits the library version has the same shape, and a custom curve is usually a config callback, not a new scheduler.\nCompleteness: A=9/10, B=5/10, C=n/a (investigation, decides nothing)\nPros / cons:\nA) Library hooks + custom curve (recommended)\n \u2705 One scheduler the library already tests; the curve becomes a per-worker config function (human: ~1 day / CC: ~20 min)\n \u2705 Attempt-count persistence, max attempts and dead-letter usually come along for free with the hooks\n \u274c If the hook API cannot accept an arbitrary curve function, that is a new fact and R1 reopens\nB) Custom inline scheduler (as planned)\n \u2705 Total control of delay math and logging, independent of the library's release cycle\n \u2705 No dependency on hook semantics nobody has read yet (human: ~3 days / CC: ~1 hr)\n \u274c Five hand-rolled schedulers to keep in sync, plus reimplementing attempt persistence and terminal handling\nC) Investigate first: bounded read of the hook API\n \u2705 Settles \"can the hooks take a custom curve\" with evidence before committing either way\n \u2705 Cheap: ~30 min human / ~3 min CC reading the hook signature and docs\n \u274c Decides nothing; R1 stays pending and the review pauses on this question\nNet: You are trading a library dependency you already carry for five copies of scheduler code you would own forever.", + "header": "Retry mechanism", + "multiSelect": false, + "options": [ + { + "label": "Library hooks + custom curve (recommended)", + "description": "Use the job library's built-in retry hooks; supply each worker's backoff curve as a callback/config. One scheduler the library already tests. Attempt persistence, max attempts and dead-letter usually included. Human ~1 day / CC ~20 min. Risk: if the hook API cannot take an arbitrary curve, R1 reopens. Completeness 9/10." + }, + { + "label": "Custom inline scheduler (as planned)", + "description": "Roll the exponential-backoff scheduler inline in each of the 5 workers as PLAN.md:7-9 proposes. Total control of delay math and logging. Human ~3 days / CC ~1 hr. Cost: five schedulers to keep in sync, plus attempt persistence and terminal handling rebuilt by hand. Completeness 5/10." + }, + { + "label": "Investigate hook API first", + "description": "Bounded ~30 min human / ~3 min CC read of the library's retry hook signature and docs, then return to this question. Approves nothing; R1 stays pending; R2\u2013R6 unchanged." + } + ] + } + ], + "answered": true, + "failed": false, + "answers": { + "D1 \u2014 Reuse the job library's retry hooks or roll a custom scheduler?\nProject/branch/task: main branch, retry-framework plan; adding retries to 5 background workers.\nELI10: The job library you already use has retry hooks built in, and your plan says your custom version would be \"the same shape.\" Building your own copy inside each worker means five hand-written schedulers to keep correct, versus configuring one curve the library already knows how to run. The plan's reason for going custom is \"full control over the curve,\" and most retry hook APIs give you that through a backoff callback.\nStakes if we pick wrong: five bespoke schedulers drift apart, each grows its own bugs (no jitter, no cap, retry storms), and nobody at 3am knows which curve a given worker actually runs.\nRecommendation: A because the plan admits the library version has the same shape, and a custom curve is usually a config callback, not a new scheduler.\nCompleteness: A=9/10, B=5/10, C=n/a (investigation, decides nothing)\nPros / cons:\nA) Library hooks + custom curve (recommended)\n \u2705 One scheduler the library already tests; the curve becomes a per-worker config function (human: ~1 day / CC: ~20 min)\n \u2705 Attempt-count persistence, max attempts and dead-letter usually come along for free with the hooks\n \u274c If the hook API cannot accept an arbitrary curve function, that is a new fact and R1 reopens\nB) Custom inline scheduler (as planned)\n \u2705 Total control of delay math and logging, independent of the library's release cycle\n \u2705 No dependency on hook semantics nobody has read yet (human: ~3 days / CC: ~1 hr)\n \u274c Five hand-rolled schedulers to keep in sync, plus reimplementing attempt persistence and terminal handling\nC) Investigate first: bounded read of the hook API\n \u2705 Settles \"can the hooks take a custom curve\" with evidence before committing either way\n \u2705 Cheap: ~30 min human / ~3 min CC reading the hook signature and docs\n \u274c Decides nothing; R1 stays pending and the review pauses on this question\nNet: You are trading a library dependency you already carry for five copies of scheduler code you would own forever.": "Library hooks + custom curve (recommended)" + }, + "unansweredQuestionIndices": [], + "answeredAt": "2026-09-29T19:51:49.159Z" + }, + { + "sessionId": "0117efe3-a002-42c1-aa59-d45032643e13", + "toolUseId": "toolu_015Ri9YxuhvdexzxG5KqBTTc", + "questions": [ + { + "question": "D2 \u2014 What delivery guarantee does processWebhookJob() keep once it can retry?\nProject/branch/task: main branch, retry-framework plan; the webhook worker is one of the 5 workers gaining retries through library hooks (D1).\nELI10: Today the webhook worker sends each event at most once: if the send fails or times out, the event is dropped, never duplicated. A retry cannot tell \"the request never arrived\" apart from \"it arrived but the response got lost,\" so any retry after a timeout can deliver the same event twice. Adding retries silently flips the guarantee from at-most-once to at-least-once. That is a contract change your webhook receivers depend on, and the plan does not name it.\nStakes if we pick wrong: receivers that are not idempotent process duplicate events (double emails, double charges, double state transitions); or, if we keep dropping on ambiguity, the retry framework never fixes the webhook worker's lost events.\nRecommendation: B because losing events is usually worse than duplicates, and a stable idempotency key makes duplicates safe for receivers; this is still a contract call you know better than the review does.\nNote: options differ in kind, not coverage \u2014 no completeness score.\nPros / cons:\nA) Keep at-most-once: retry only provably-unsent failures\n \u2705 No duplicate deliveries ever; existing receivers keep working with no change on their side\n \u2705 Still recovers the clear cases: connection refused, DNS failure, local enqueue error (human: ~1 day / CC: ~30 min)\n \u274c Timeouts and 5xx-after-send still drop events, so the biggest source of loss stays; needs per-attempt failure classification\nB) Move to at-least-once with a stable idempotency key (recommended)\n \u2705 Every event eventually reaches the receiver; retries after timeouts are safe because the event id stays constant across attempts\n \u2705 Matches Stripe, Shopify and Svix practice; receivers dedupe on the key (human: ~1.5 days / CC: ~30 min incl. docs)\n \u274c Contract change: receivers must dedupe; needs a documented header, changelog entry and migration note for existing receivers\nC) Exclude processWebhookJob() from retries\n \u2705 Zero semantic change for receivers; the other 4 workers still get retries\n \u2705 Smallest diff and no receiver communication (human: ~1 hr / CC: ~5 min)\n \u274c The webhook worker keeps losing events on every transient failure, which is likely why the plan touched it\nNet: Never-duplicate-but-lossy, never-lossy-but-receivers-must-dedupe, or leave the webhook worker exactly as it is.", + "header": "Webhook delivery", + "multiSelect": false, + "options": [ + { + "label": "Keep at-most-once (retry only pre-send failures)", + "description": "Webhook worker retries only failures provably raised before the request was written (connection refused, DNS, local error). Timeouts and 5xx-after-send still drop the event. No duplicates; receivers unchanged. Needs per-attempt failure classification. Human ~1 day / CC ~30 min." + }, + { + "label": "At-least-once + idempotency key (recommended)", + "description": "Webhook worker retries all transient failures; every delivery carries a stable event id / idempotency key header constant across attempts. Documented contract change with changelog and receiver migration note. Receivers dedupe on the key. Human ~1.5 days / CC ~30 min." + }, + { + "label": "Exclude webhook worker from retries", + "description": "`processWebhookJob()` keeps today's at-most-once, no-retry behavior; the other 4 workers retry via library hooks per D1. Smallest diff, no receiver impact, webhook events still lost on transient failure. Human ~1 hr / CC ~5 min." + } + ] + } + ], + "answered": true, + "failed": false, + "answers": { + "D2 \u2014 What delivery guarantee does processWebhookJob() keep once it can retry?\nProject/branch/task: main branch, retry-framework plan; the webhook worker is one of the 5 workers gaining retries through library hooks (D1).\nELI10: Today the webhook worker sends each event at most once: if the send fails or times out, the event is dropped, never duplicated. A retry cannot tell \"the request never arrived\" apart from \"it arrived but the response got lost,\" so any retry after a timeout can deliver the same event twice. Adding retries silently flips the guarantee from at-most-once to at-least-once. That is a contract change your webhook receivers depend on, and the plan does not name it.\nStakes if we pick wrong: receivers that are not idempotent process duplicate events (double emails, double charges, double state transitions); or, if we keep dropping on ambiguity, the retry framework never fixes the webhook worker's lost events.\nRecommendation: B because losing events is usually worse than duplicates, and a stable idempotency key makes duplicates safe for receivers; this is still a contract call you know better than the review does.\nNote: options differ in kind, not coverage \u2014 no completeness score.\nPros / cons:\nA) Keep at-most-once: retry only provably-unsent failures\n \u2705 No duplicate deliveries ever; existing receivers keep working with no change on their side\n \u2705 Still recovers the clear cases: connection refused, DNS failure, local enqueue error (human: ~1 day / CC: ~30 min)\n \u274c Timeouts and 5xx-after-send still drop events, so the biggest source of loss stays; needs per-attempt failure classification\nB) Move to at-least-once with a stable idempotency key (recommended)\n \u2705 Every event eventually reaches the receiver; retries after timeouts are safe because the event id stays constant across attempts\n \u2705 Matches Stripe, Shopify and Svix practice; receivers dedupe on the key (human: ~1.5 days / CC: ~30 min incl. docs)\n \u274c Contract change: receivers must dedupe; needs a documented header, changelog entry and migration note for existing receivers\nC) Exclude processWebhookJob() from retries\n \u2705 Zero semantic change for receivers; the other 4 workers still get retries\n \u2705 Smallest diff and no receiver communication (human: ~1 hr / CC: ~5 min)\n \u274c The webhook worker keeps losing events on every transient failure, which is likely why the plan touched it\nNet: Never-duplicate-but-lossy, never-lossy-but-receivers-must-dedupe, or leave the webhook worker exactly as it is.": "Keep at-most-once (retry only pre-send failures)" + }, + "unansweredQuestionIndices": [], + "answeredAt": "2026-09-29T19:54:01.993Z" + }, + { + "sessionId": "0117efe3-a002-42c1-aa59-d45032643e13", + "toolUseId": "toolu_0132i9qkrzXUVyovu4GZxADn", + "questions": [ + { + "question": "D3 \u2014 When a job runs out of retries, where does it go?\nProject/branch/task: main branch, retry-framework plan; retry bounds for all 5 workers running through library hooks (D1).\nELI10: Right now the plan describes the curve between retries but never says how many retries there are or what happens to a job that keeps failing. Without a limit, a poisoned job retries forever and eats worker capacity. With a limit but no landing spot, the job disappears with one log line nobody reads. A dead-letter store keeps the failed job, its payload reference and its last error so someone can inspect and replay it.\nStakes if we pick wrong: either an infinite-retry job starves the queue, or real work silently vanishes after the last attempt and the first sign is a customer asking where their data went.\nRecommendation: A because a dead-letter store plus an alert is a few dozen lines with library hooks, and it turns \"job vanished\" into \"job parked, here is why.\"\nCompleteness: A=10/10, B=5/10, C=3/10\nPros / cons:\nA) Bounded attempts + dead-letter store + alert (recommended)\n \u2705 Exhausted or fatal jobs are kept with last error and attempt history; operators can inspect and replay (human: ~1 day / CC: ~20 min)\n \u2705 Metric and alert on dead-letter growth turns a silent failure into a page at the right time\n \u274c One more table or queue to own, plus a small replay path to build and test\nB) Bounded attempts, log and drop\n \u2705 Simplest bound: `maxAttempts` default 5 per worker, one error log on exhaustion (human: ~2 hr / CC: ~5 min)\n \u2705 No new storage; nothing to operate\n \u274c Exhausted jobs are gone; recovery means replaying from upstream sources by hand, if that is even possible\nC) Leave to library defaults\n \u2705 Zero plan text and zero decision now\n \u2705 Whatever the library does is at least consistent across the 5 workers\n \u274c Nobody knows the limit or the terminal behavior until an incident teaches them; 3am failure mode\nNet: You are trading one small dead-letter store for never having to ask \"where did that job go.\"", + "header": "Retry exhaustion", + "multiSelect": false, + "options": [ + { + "label": "Bounded + dead-letter + alert (recommended)", + "description": "`maxAttempts` default 5 with per-worker override. On exhaustion or fatal error the job lands in a dead-letter store (table or queue) with last error, attempt history and payload reference. Metric and alert on dead-letter growth. Manual replay path. Human ~1 day / CC ~20 min. Completeness 10/10." + }, + { + "label": "Bounded, log and drop", + "description": "`maxAttempts` default 5 with per-worker override. On exhaustion, log at error level with the last error and drop the job. No new storage, no replay. Human ~2 hr / CC ~5 min. Completeness 5/10." + }, + { + "label": "Library defaults, unspecified", + "description": "Do not write attempt limits or terminal behavior into the plan; accept whatever the library does by default. Completeness 3/10." + } + ] + } + ], + "answered": true, + "failed": false, + "answers": { + "D3 \u2014 When a job runs out of retries, where does it go?\nProject/branch/task: main branch, retry-framework plan; retry bounds for all 5 workers running through library hooks (D1).\nELI10: Right now the plan describes the curve between retries but never says how many retries there are or what happens to a job that keeps failing. Without a limit, a poisoned job retries forever and eats worker capacity. With a limit but no landing spot, the job disappears with one log line nobody reads. A dead-letter store keeps the failed job, its payload reference and its last error so someone can inspect and replay it.\nStakes if we pick wrong: either an infinite-retry job starves the queue, or real work silently vanishes after the last attempt and the first sign is a customer asking where their data went.\nRecommendation: A because a dead-letter store plus an alert is a few dozen lines with library hooks, and it turns \"job vanished\" into \"job parked, here is why.\"\nCompleteness: A=10/10, B=5/10, C=3/10\nPros / cons:\nA) Bounded attempts + dead-letter store + alert (recommended)\n \u2705 Exhausted or fatal jobs are kept with last error and attempt history; operators can inspect and replay (human: ~1 day / CC: ~20 min)\n \u2705 Metric and alert on dead-letter growth turns a silent failure into a page at the right time\n \u274c One more table or queue to own, plus a small replay path to build and test\nB) Bounded attempts, log and drop\n \u2705 Simplest bound: `maxAttempts` default 5 per worker, one error log on exhaustion (human: ~2 hr / CC: ~5 min)\n \u2705 No new storage; nothing to operate\n \u274c Exhausted jobs are gone; recovery means replaying from upstream sources by hand, if that is even possible\nC) Leave to library defaults\n \u2705 Zero plan text and zero decision now\n \u2705 Whatever the library does is at least consistent across the 5 workers\n \u274c Nobody knows the limit or the terminal behavior until an incident teaches them; 3am failure mode\nNet: You are trading one small dead-letter store for never having to ask \"where did that job go.\"": "Bounded + dead-letter + alert (recommended)" + }, + "unansweredQuestionIndices": [], + "answeredAt": "2026-09-29T19:55:16.802Z" + }, + { + "sessionId": "0117efe3-a002-42c1-aa59-d45032643e13", + "toolUseId": "toolu_01GMRg1vxVPnVCefhfuCSKEw", + "questions": [ + { + "question": "D4 \u2014 Should retry delays be randomized (jitter)?\nProject/branch/task: main branch, retry-framework plan; the backoff curve each worker supplies to the library hooks (D1).\nELI10: When many jobs fail at the same moment because a shared dependency went down, a pure exponential curve makes them all retry at the same moments too, so the recovering dependency gets hit by a wave on every step. Jitter randomizes each job's delay so the retries spread out. It is one line inside the curve callback each worker already supplies.\nStakes if we pick wrong: synchronized retry waves knock a recovering dependency back over (the classic thundering herd); or, with jitter, per-job retry timing becomes slightly less predictable and tests need a seeded random source.\nRecommendation: A because full jitter gives the least contention in AWS's published analysis and costs one line in a callback you are already writing.\nNote: options differ in kind, not coverage \u2014 no completeness score.\nPros / cons:\nA) Full jitter: random(0, exponentialDelay) (recommended)\n \u2705 Best spread of retries and lowest total contention after a shared outage (AWS Builders' Library)\n \u2705 One line inside the D1 curve callback; RNG injected so tests stay deterministic (human: ~1 hr / CC: ~5 min)\n \u274c An individual retry can fire almost immediately; minimum wait is not guaranteed\nB) Equal jitter: half fixed, half random\n \u2705 Guarantees a minimum wait of half the exponential delay while still spreading retries\n \u2705 Same one-line cost and same injectable RNG as full jitter (human: ~1 hr / CC: ~5 min)\n \u274c Slightly more contention than full jitter in the same analysis, and one more parameter to explain\nC) No jitter: deterministic curve\n \u2705 Fully deterministic; trivial to reason about and to assert exact delays in tests\n \u2705 Zero extra code beyond the exponential curve\n \u274c Every job that failed together retries together; retry storms on recovery are the expected outcome\nNet: One random() call now versus a synchronized retry wave the first time a dependency has a bad hour.", + "header": "Jitter", + "multiSelect": false, + "options": [ + { + "label": "Full jitter (recommended)", + "description": "delay = random(0, exponentialDelay) inside each worker's curve callback. Best spread, lowest contention. RNG injectable so tests are deterministic. Human ~1 hr / CC ~5 min." + }, + { + "label": "Equal jitter", + "description": "delay = exponentialDelay/2 + random(0, exponentialDelay/2). Guarantees a minimum wait; slightly more contention than full jitter. RNG injectable. Human ~1 hr / CC ~5 min." + }, + { + "label": "No jitter", + "description": "Deterministic exponential curve, no randomization. Simplest to test; retries synchronize after a shared outage." + } + ] + } + ], + "answered": true, + "failed": false, + "answers": { + "D4 \u2014 Should retry delays be randomized (jitter)?\nProject/branch/task: main branch, retry-framework plan; the backoff curve each worker supplies to the library hooks (D1).\nELI10: When many jobs fail at the same moment because a shared dependency went down, a pure exponential curve makes them all retry at the same moments too, so the recovering dependency gets hit by a wave on every step. Jitter randomizes each job's delay so the retries spread out. It is one line inside the curve callback each worker already supplies.\nStakes if we pick wrong: synchronized retry waves knock a recovering dependency back over (the classic thundering herd); or, with jitter, per-job retry timing becomes slightly less predictable and tests need a seeded random source.\nRecommendation: A because full jitter gives the least contention in AWS's published analysis and costs one line in a callback you are already writing.\nNote: options differ in kind, not coverage \u2014 no completeness score.\nPros / cons:\nA) Full jitter: random(0, exponentialDelay) (recommended)\n \u2705 Best spread of retries and lowest total contention after a shared outage (AWS Builders' Library)\n \u2705 One line inside the D1 curve callback; RNG injected so tests stay deterministic (human: ~1 hr / CC: ~5 min)\n \u274c An individual retry can fire almost immediately; minimum wait is not guaranteed\nB) Equal jitter: half fixed, half random\n \u2705 Guarantees a minimum wait of half the exponential delay while still spreading retries\n \u2705 Same one-line cost and same injectable RNG as full jitter (human: ~1 hr / CC: ~5 min)\n \u274c Slightly more contention than full jitter in the same analysis, and one more parameter to explain\nC) No jitter: deterministic curve\n \u2705 Fully deterministic; trivial to reason about and to assert exact delays in tests\n \u2705 Zero extra code beyond the exponential curve\n \u274c Every job that failed together retries together; retry storms on recovery are the expected outcome\nNet: One random() call now versus a synchronized retry wave the first time a dependency has a bad hour.": "Full jitter (recommended)" + }, + "unansweredQuestionIndices": [], + "answeredAt": "2026-09-29T19:56:30.842Z" + }, + { + "sessionId": "0117efe3-a002-42c1-aa59-d45032643e13", + "toolUseId": "toolu_01XLVRABVWXpSZRBNni2esSC", + "questions": [ + { + "question": "D5 \u2014 Should the backoff delay have a ceiling?\nProject/branch/task: main branch, retry-framework plan; the curve parameters each worker passes to the library hooks (D1, jittered per D4).\nELI10: Exponential backoff doubles the wait after each failure. That is fine for 5 attempts, but D3 lets each worker raise its attempt count, and a worker set to 15 attempts from a 1 second base would wait about 4.5 hours before its last try; at 20 attempts it would wait 6 days. A cap says \"never wait longer than X between attempts,\" so the curve grows and then flattens. It is one min() call in the callback.\nStakes if we pick wrong: without a cap, a worker with a higher attempt count silently turns into a multi-day wait that looks like a stuck job; with a cap, one more number to document per worker.\nRecommendation: A because the cap is one min() and it makes \"how long can this job be delayed\" a question with an answer.\nCompleteness: A=9/10, B=4/10\nPros / cons:\nA) Cap each delay: default 10 min, per-worker override (recommended)\n \u2705 Worst-case wait between attempts is bounded and documented for every worker (human: ~1 hr / CC: ~5 min)\n \u2705 Also pins the curve defaults (base 1 s, multiplier 2) so all 5 workers start from the same documented numbers\n \u274c One more config value per worker to document and keep sane alongside maxAttempts\nB) No cap\n \u2705 Zero code; the curve is exactly the exponential the plan describes\n \u2705 Fewer knobs to explain\n \u274c Any worker that raises maxAttempts past ~12 gets hour-to-day waits nobody intended\nNet: One min() now versus a job that looks stuck for six days the first time someone bumps an attempt count.", + "header": "Delay cap", + "multiSelect": false, + "options": [ + { + "label": "Cap each delay (recommended)", + "description": "delay = min(jitteredExponential, maxDelay). `maxDelay` default 10 minutes with per-worker override. Curve defaults documented: base 1 s, multiplier 2, per-worker override. Human ~1 hr / CC ~5 min. Completeness 9/10." + }, + { + "label": "No cap", + "description": "Raw exponential curve with no ceiling. Zero code, fewer knobs; high attempt counts produce hour-to-day waits. Completeness 4/10." + } + ] + } + ], + "answered": true, + "failed": false, + "answers": { + "D5 \u2014 Should the backoff delay have a ceiling?\nProject/branch/task: main branch, retry-framework plan; the curve parameters each worker passes to the library hooks (D1, jittered per D4).\nELI10: Exponential backoff doubles the wait after each failure. That is fine for 5 attempts, but D3 lets each worker raise its attempt count, and a worker set to 15 attempts from a 1 second base would wait about 4.5 hours before its last try; at 20 attempts it would wait 6 days. A cap says \"never wait longer than X between attempts,\" so the curve grows and then flattens. It is one min() call in the callback.\nStakes if we pick wrong: without a cap, a worker with a higher attempt count silently turns into a multi-day wait that looks like a stuck job; with a cap, one more number to document per worker.\nRecommendation: A because the cap is one min() and it makes \"how long can this job be delayed\" a question with an answer.\nCompleteness: A=9/10, B=4/10\nPros / cons:\nA) Cap each delay: default 10 min, per-worker override (recommended)\n \u2705 Worst-case wait between attempts is bounded and documented for every worker (human: ~1 hr / CC: ~5 min)\n \u2705 Also pins the curve defaults (base 1 s, multiplier 2) so all 5 workers start from the same documented numbers\n \u274c One more config value per worker to document and keep sane alongside maxAttempts\nB) No cap\n \u2705 Zero code; the curve is exactly the exponential the plan describes\n \u2705 Fewer knobs to explain\n \u274c Any worker that raises maxAttempts past ~12 gets hour-to-day waits nobody intended\nNet: One min() now versus a job that looks stuck for six days the first time someone bumps an attempt count.": "Cap each delay (recommended)" + }, + "unansweredQuestionIndices": [], + "answeredAt": "2026-09-29T19:57:33.114Z" + }, + { + "sessionId": "0117efe3-a002-42c1-aa59-d45032643e13", + "toolUseId": "toolu_01SrWycxMxofjLp1hxkciPj9", + "questions": [ + { + "question": "D6 \u2014 Which errors should the four non-webhook workers retry, and which go straight to dead-letter?\nProject/branch/task: main branch, retry-framework plan; error handling inside the 4 non-webhook workers' library retry hooks (D1). The webhook worker's rule is already fixed by D2.\nELI10: Not every failure is worth retrying. A timeout or a \"service busy\" reply will likely pass on the next try. A validation error, a missing record or a bug that throws will fail the same way five times in a row, burning worker time and delaying the dead-letter record (D3) by the whole backoff curve. Classifying errors sends the hopeless ones to dead-letter immediately and spends retries only on the ones that can recover. The open question is what to do with an error nobody has classified yet.\nStakes if we pick wrong: either a bug retries five times per job across a whole queue before anyone sees it, or a transient error that nobody thought to list dead-letters real work on its first failure.\nRecommendation: A because classification is a short list per worker, and defaulting unknown errors to retryable never loses work: the worst case is five wasted attempts, not a dropped job.\nCompleteness: A=10/10, B=5/10, C=8/10\nPros / cons:\nA) Classify; unknown errors retry (recommended)\n \u2705 Hopeless errors (validation, 4xx, missing record, TypeError) land in dead-letter on attempt 1 with the real cause visible (human: ~half day / CC: ~15 min)\n \u2705 Unlisted errors still retry, so a forgotten transient class costs attempts, never data\n \u274c Each worker maintains a small error-class list, and a new fatal class retries needlessly until someone adds it\nB) Retry everything until maxAttempts\n \u2705 No lists to maintain; identical behavior in all 4 workers (human: ~0 / CC: ~0)\n \u2705 Impossible to misclassify a transient error as fatal\n \u274c A deploy with a bug retries every affected job 5 times over the full curve before dead-lettering; queue capacity burns and diagnosis is delayed\nC) Classify; unknown errors are fatal\n \u2705 Zero wasted attempts on anything not explicitly known to be transient (human: ~half day / CC: ~15 min)\n \u2705 Dead-letter fills fast, so new error classes surface quickly\n \u274c Any transient error missing from the list dead-letters real work on its first failure, which is the exact loss the retry framework exists to prevent\nNet: A short list per worker plus a safe default, versus either wasted retries on bugs or lost work on unlisted transients.", + "header": "Error classes", + "multiSelect": false, + "options": [ + { + "label": "Classify; unknown \u2192 retryable (recommended)", + "description": "Each of the 4 workers declares retryable classes (timeout, connection error, 429/503, deadlock/serialization failure) and fatal classes (validation error, 4xx other than 429, missing record, programming error). Fatal \u2192 dead-letter immediately without consuming attempts. Unclassified errors retry. Human ~half day / CC ~15 min. Completeness 10/10." + }, + { + "label": "Retry everything to maxAttempts", + "description": "No classification. Every error consumes an attempt until `maxAttempts`, then dead-letter per D3. Zero code; bugs retry 5 times per job. Completeness 5/10." + }, + { + "label": "Classify; unknown \u2192 fatal", + "description": "Same declared classes as A, but unclassified errors go to dead-letter immediately. No wasted attempts; unlisted transient errors lose work on first failure. Human ~half day / CC ~15 min. Completeness 8/10." + } + ] + } + ], + "answered": true, + "failed": false, + "answers": { + "D6 \u2014 Which errors should the four non-webhook workers retry, and which go straight to dead-letter?\nProject/branch/task: main branch, retry-framework plan; error handling inside the 4 non-webhook workers' library retry hooks (D1). The webhook worker's rule is already fixed by D2.\nELI10: Not every failure is worth retrying. A timeout or a \"service busy\" reply will likely pass on the next try. A validation error, a missing record or a bug that throws will fail the same way five times in a row, burning worker time and delaying the dead-letter record (D3) by the whole backoff curve. Classifying errors sends the hopeless ones to dead-letter immediately and spends retries only on the ones that can recover. The open question is what to do with an error nobody has classified yet.\nStakes if we pick wrong: either a bug retries five times per job across a whole queue before anyone sees it, or a transient error that nobody thought to list dead-letters real work on its first failure.\nRecommendation: A because classification is a short list per worker, and defaulting unknown errors to retryable never loses work: the worst case is five wasted attempts, not a dropped job.\nCompleteness: A=10/10, B=5/10, C=8/10\nPros / cons:\nA) Classify; unknown errors retry (recommended)\n \u2705 Hopeless errors (validation, 4xx, missing record, TypeError) land in dead-letter on attempt 1 with the real cause visible (human: ~half day / CC: ~15 min)\n \u2705 Unlisted errors still retry, so a forgotten transient class costs attempts, never data\n \u274c Each worker maintains a small error-class list, and a new fatal class retries needlessly until someone adds it\nB) Retry everything until maxAttempts\n \u2705 No lists to maintain; identical behavior in all 4 workers (human: ~0 / CC: ~0)\n \u2705 Impossible to misclassify a transient error as fatal\n \u274c A deploy with a bug retries every affected job 5 times over the full curve before dead-lettering; queue capacity burns and diagnosis is delayed\nC) Classify; unknown errors are fatal\n \u2705 Zero wasted attempts on anything not explicitly known to be transient (human: ~half day / CC: ~15 min)\n \u2705 Dead-letter fills fast, so new error classes surface quickly\n \u274c Any transient error missing from the list dead-letters real work on its first failure, which is the exact loss the retry framework exists to prevent\nNet: A short list per worker plus a safe default, versus either wasted retries on bugs or lost work on unlisted transients.": "Classify; unknown \u2192 retryable (recommended)" + }, + "unansweredQuestionIndices": [], + "answeredAt": "2026-09-29T19:58:40.484Z" + }, + { + "sessionId": "0117efe3-a002-42c1-aa59-d45032643e13", + "toolUseId": "toolu_013bXU7e1WUvhP4r6Nh2cigC", + "questions": [ + { + "question": "D7 \u2014 One shared retry-policy module, or five copies and \"refactor later\"?\nProject/branch/task: main branch, retry-framework plan; how the 5 workers carry the behavior approved in D3\u2013D6.\nELI10: After D1 the library does the scheduling, but every worker still has to hand it the same four things: a jittered, capped curve, an error classifier, a dead-letter handoff and an attempt log line. The plan copies that block into five files and promises to clean up later. \"Later\" for copy-pasted retry code usually means the fifth copy drifts (no cap, wrong jitter) and nobody notices until an incident. The alternative is one small module that each worker configures with its own numbers and error lists.\nStakes if we pick wrong: five curves that silently disagree, five places to fix the next retry bug, and five test suites that each cover a slightly different subset; or, with a shared module, one bug that hits all five workers at once (mitigated by the module's own tests).\nRecommendation: A because the behavior is identical by construction (D3\u2013D6 fixed it), the module is under 100 lines, and it removes more lines than it adds while making the retry rules testable once.\nCompleteness: A=10/10, B=4/10, C=7/10\nPros / cons:\nA) One shared retry-policy module (recommended)\n \u2705 Curve, classifier, dead-letter handoff, attempt log and config validation are tested once and behave the same in all 5 workers (human: ~1 day / CC: ~20 min)\n \u2705 Estimated 15\u201390 implementation lines saved; the helper's test suite replaces five near-duplicate suites\n \u274c A bug in the module reaches all 5 workers; the module's own tests are the guard\nB) Five inline copies, refactor later (as planned)\n \u2705 No shared dependency between workers; each can be changed in isolation (human: ~1.5 days / CC: ~30 min)\n \u2705 Matches the plan text exactly; nothing new to name or place\n \u274c Five copies to keep in sync and five test suites to write; \"later\" rarely arrives for retry glue\nC) Extract the curve builder only\n \u2705 The math most likely to drift (jitter + cap) lives in one place (human: ~1 day / CC: ~15 min)\n \u2705 Smaller shared surface than A\n \u274c Classifier, dead-letter handoff and log line are still copied five times, so most of the duplication and its tests remain\nNet: One under-100-line module now, or five copies plus a promise.", + "header": "Shared module", + "multiSelect": false, + "options": [ + { + "label": "One shared retry-policy module (recommended)", + "description": "Small module: buildBackoff (base/multiplier/maxDelay/rng), classify (per-worker retryable/fatal lists), toDeadLetter (persist + metric), logAttempt, and startup config validation. All 5 workers use it with their own values. Human ~1 day / CC ~20 min. Completeness 10/10." + }, + { + "label": "Five inline copies (as planned)", + "description": "Each worker carries its own curve, classifier, dead-letter handoff and log line; refactor deferred. Human ~1.5 days / CC ~30 min. Completeness 4/10." + }, + { + "label": "Extract curve builder only", + "description": "Shared buildBackoff (jitter + cap) only; classifier, dead-letter handoff and log line stay inline in each of the 5 workers. Human ~1 day / CC ~15 min. Completeness 7/10." + } + ] + } + ], + "answered": true, + "failed": false, + "answers": { + "D7 \u2014 One shared retry-policy module, or five copies and \"refactor later\"?\nProject/branch/task: main branch, retry-framework plan; how the 5 workers carry the behavior approved in D3\u2013D6.\nELI10: After D1 the library does the scheduling, but every worker still has to hand it the same four things: a jittered, capped curve, an error classifier, a dead-letter handoff and an attempt log line. The plan copies that block into five files and promises to clean up later. \"Later\" for copy-pasted retry code usually means the fifth copy drifts (no cap, wrong jitter) and nobody notices until an incident. The alternative is one small module that each worker configures with its own numbers and error lists.\nStakes if we pick wrong: five curves that silently disagree, five places to fix the next retry bug, and five test suites that each cover a slightly different subset; or, with a shared module, one bug that hits all five workers at once (mitigated by the module's own tests).\nRecommendation: A because the behavior is identical by construction (D3\u2013D6 fixed it), the module is under 100 lines, and it removes more lines than it adds while making the retry rules testable once.\nCompleteness: A=10/10, B=4/10, C=7/10\nPros / cons:\nA) One shared retry-policy module (recommended)\n \u2705 Curve, classifier, dead-letter handoff, attempt log and config validation are tested once and behave the same in all 5 workers (human: ~1 day / CC: ~20 min)\n \u2705 Estimated 15\u201390 implementation lines saved; the helper's test suite replaces five near-duplicate suites\n \u274c A bug in the module reaches all 5 workers; the module's own tests are the guard\nB) Five inline copies, refactor later (as planned)\n \u2705 No shared dependency between workers; each can be changed in isolation (human: ~1.5 days / CC: ~30 min)\n \u2705 Matches the plan text exactly; nothing new to name or place\n \u274c Five copies to keep in sync and five test suites to write; \"later\" rarely arrives for retry glue\nC) Extract the curve builder only\n \u2705 The math most likely to drift (jitter + cap) lives in one place (human: ~1 day / CC: ~15 min)\n \u2705 Smaller shared surface than A\n \u274c Classifier, dead-letter handoff and log line are still copied five times, so most of the duplication and its tests remain\nNet: One under-100-line module now, or five copies plus a promise.": "One shared retry-policy module (recommended)" + }, + "unansweredQuestionIndices": [], + "answeredAt": "2026-09-29T20:00:17.086Z" + }, + { + "sessionId": "0117efe3-a002-42c1-aa59-d45032643e13", + "toolUseId": "toolu_01Ly1JT6sRQHG4ZVpQ7SUETD", + "questions": [ + { + "question": "D8 \u2014 How do we prove processWebhookJob() still sends each event at most once?\nProject/branch/task: main branch, retry-framework plan; regression coverage for the rewritten webhook worker (D2 fixed the behavior to keep).\nELI10: The webhook worker is being rewritten and it carries a promise to receivers: an event is never sent twice. D2 kept that promise while adding retries for failures that happen before anything is sent. A rewrite with no test for the promise means the first duplicate email or double charge is found by a customer. The test is straightforward: a fake receiver counts sends, and we assert the count is exactly one across every failure pattern. The question is how deep to go: assertions against the worker alone, a run through the real library hooks, or both.\nStakes if we pick wrong: a retry path nobody tested sends duplicates to non-idempotent receivers, or a hook wiring mistake means pre-send failures never actually retry and the framework quietly does nothing for webhooks.\nRecommendation: A because the unit layer pins each failure class cheaply and the integration layer is the only thing that catches hook wiring and attempt persistence, which is where retry bugs actually live.\nCompleteness: A=10/10, B=7/10, C=7/10\nPros / cons:\nA) Unit + integration through the library hooks (recommended)\n \u2705 Every failure class (pre-send, timeout, 5xx, reset, success) asserted in isolation with a recording fake transport (human: ~1 day / CC: ~20 min)\n \u2705 One end-to-end run through the real hooks with a fake receiver catches wiring and attempt-persistence bugs the unit layer cannot see\n \u274c Two test layers to maintain; the integration test needs the library's test harness or an in-process queue\nB) Unit tests only\n \u2705 Fast, deterministic, no queue infrastructure in the test run (human: ~half day / CC: ~10 min)\n \u2705 Pins the classifier and the send-count contract per failure class\n \u274c Never exercises the real hook registration, so a miswired hook passes tests and never retries in production\nC) Integration test only\n \u2705 Exercises the real path receivers depend on (human: ~half day / CC: ~10 min)\n \u2705 Fewer tests to write\n \u274c Slower, and a failure tells you \"something duplicated\" without pointing at which failure class; edge classes get skipped for time\nNet: Cheap isolated assertions plus one real-path run, versus trusting either layer alone to protect a promise made to external receivers.", + "header": "Webhook regression", + "multiSelect": false, + "options": [ + { + "label": "Unit + integration (recommended)", + "description": "Unit: fake transport records every send; assert exactly 1 send after pre-send retries, 0 further sends after timeout/5xx/reset with a dead-letter entry, 1 send on success. Integration: real library hooks + fake receiver, same assertions, plus attempt count survives a simulated worker restart. Human ~1 day / CC ~20 min. Completeness 10/10." + }, + { + "label": "Unit tests only", + "description": "The unit assertions from A against the worker with a fake transport; no run through the real library hooks. Human ~half day / CC ~10 min. Completeness 7/10." + }, + { + "label": "Integration test only", + "description": "The integration run from A only; no isolated per-failure-class assertions. Human ~half day / CC ~10 min. Completeness 7/10." + } + ] + } + ], + "answered": true, + "failed": false, + "answers": { + "D8 \u2014 How do we prove processWebhookJob() still sends each event at most once?\nProject/branch/task: main branch, retry-framework plan; regression coverage for the rewritten webhook worker (D2 fixed the behavior to keep).\nELI10: The webhook worker is being rewritten and it carries a promise to receivers: an event is never sent twice. D2 kept that promise while adding retries for failures that happen before anything is sent. A rewrite with no test for the promise means the first duplicate email or double charge is found by a customer. The test is straightforward: a fake receiver counts sends, and we assert the count is exactly one across every failure pattern. The question is how deep to go: assertions against the worker alone, a run through the real library hooks, or both.\nStakes if we pick wrong: a retry path nobody tested sends duplicates to non-idempotent receivers, or a hook wiring mistake means pre-send failures never actually retry and the framework quietly does nothing for webhooks.\nRecommendation: A because the unit layer pins each failure class cheaply and the integration layer is the only thing that catches hook wiring and attempt persistence, which is where retry bugs actually live.\nCompleteness: A=10/10, B=7/10, C=7/10\nPros / cons:\nA) Unit + integration through the library hooks (recommended)\n \u2705 Every failure class (pre-send, timeout, 5xx, reset, success) asserted in isolation with a recording fake transport (human: ~1 day / CC: ~20 min)\n \u2705 One end-to-end run through the real hooks with a fake receiver catches wiring and attempt-persistence bugs the unit layer cannot see\n \u274c Two test layers to maintain; the integration test needs the library's test harness or an in-process queue\nB) Unit tests only\n \u2705 Fast, deterministic, no queue infrastructure in the test run (human: ~half day / CC: ~10 min)\n \u2705 Pins the classifier and the send-count contract per failure class\n \u274c Never exercises the real hook registration, so a miswired hook passes tests and never retries in production\nC) Integration test only\n \u2705 Exercises the real path receivers depend on (human: ~half day / CC: ~10 min)\n \u2705 Fewer tests to write\n \u274c Slower, and a failure tells you \"something duplicated\" without pointing at which failure class; edge classes get skipped for time\nNet: Cheap isolated assertions plus one real-path run, versus trusting either layer alone to protect a promise made to external receivers.": "Unit + integration (recommended)" + }, + "unansweredQuestionIndices": [], + "answeredAt": "2026-09-29T20:02:01.530Z" + }, + { + "sessionId": "0117efe3-a002-42c1-aa59-d45032643e13", + "toolUseId": "toolu_01WZ2b5HaUXKgTwZxqfKn3pd", + "questions": [ + { + "question": "D9 \u2014 Cache the dependency graph across retries now, or measure first?\nProject/branch/task: main branch, retry-framework plan; per-attempt cost inside the 5 workers running through library hooks (D1).\nELI10: The plan worries that each retry reloads the job and rebuilds its dependency graph from scratch. After D1 the reload is just the library handing the job to the worker, which happens anyway. The rebuild is real extra CPU, but only on retries, and D3 caps those at 5 per failing job. Storing the graph on the first attempt would add a write to every job, including the large majority that succeed first time, to save work on the few that fail. Nobody has measured how long the rebuild takes.\nStakes if we pick wrong: either we add a write and a staleness risk to every job to fix a cost nobody measured, or a genuinely slow rebuild keeps burning worker time on retries and we only find out under load.\nRecommendation: C because an unmeasured optimization that taxes the happy path is the wrong trade; two timing metrics make the real decision cheap and data-driven.\nNote: options differ in kind (persisted cache vs in-process memo vs measure first) \u2014 no completeness score.\nPros / cons:\nA) Persist the graph with the job on attempt 1\n \u2705 Retries never recompute; cost is paid once per job regardless of which worker instance retries (human: ~1 day / CC: ~20 min)\n \u2705 Simple to reason about once the invalidation rule (payload version) is in place\n \u274c Adds a write and stored blob to every job, including the ones that never retry; stale-graph bugs if the payload changes between attempts\nB) In-process memo (bounded LRU)\n \u2705 No persistence, no schema change; a few lines around the graph builder (human: ~2 hr / CC: ~10 min)\n \u2705 Zero cost on the happy path beyond a map insert\n \u274c Retries after a 10-minute delay usually land on a different worker instance, so the hit rate is low and unpredictable\nC) Measure first: timing metrics + p95 budget (recommended)\n \u2705 Two metrics (graph compute ms, payload bytes) per attempt tell you whether this is 2 ms or 2 s before anyone writes cache code (human: ~1 hr / CC: ~5 min)\n \u2705 No happy-path cost, no staleness risk, and the retry-policy module already logs per attempt (D7) so the hook point exists\n \u274c If the rebuild is genuinely slow, retries stay expensive until the follow-up lands\nNet: Add a write to every job to save CPU on the few that retry, or spend an hour on metrics and decide with numbers.", + "header": "Graph cache", + "multiSelect": false, + "options": [ + { + "label": "Persist the graph with the job", + "description": "Compute once on attempt 1, store the graph beside the job row, reuse on retries, invalidate when the payload version changes. Adds a write to every job. Human ~1 day / CC ~20 min." + }, + { + "label": "In-process memo (bounded LRU)", + "description": "Memoize the graph per worker instance keyed by job id + payload hash, bounded LRU. No persistence; low hit rate when retries land on another instance. Human ~2 hr / CC ~10 min." + }, + { + "label": "Measure first (recommended)", + "description": "No cache. Add per-attempt timing metrics (job load ms, graph compute ms, payload bytes) via the retry-policy module's attempt log, set a p95 budget, and revisit caching with data. Human ~1 hr / CC ~5 min." + } + ] + } + ], + "answered": true, + "failed": false, + "answers": { + "D9 \u2014 Cache the dependency graph across retries now, or measure first?\nProject/branch/task: main branch, retry-framework plan; per-attempt cost inside the 5 workers running through library hooks (D1).\nELI10: The plan worries that each retry reloads the job and rebuilds its dependency graph from scratch. After D1 the reload is just the library handing the job to the worker, which happens anyway. The rebuild is real extra CPU, but only on retries, and D3 caps those at 5 per failing job. Storing the graph on the first attempt would add a write to every job, including the large majority that succeed first time, to save work on the few that fail. Nobody has measured how long the rebuild takes.\nStakes if we pick wrong: either we add a write and a staleness risk to every job to fix a cost nobody measured, or a genuinely slow rebuild keeps burning worker time on retries and we only find out under load.\nRecommendation: C because an unmeasured optimization that taxes the happy path is the wrong trade; two timing metrics make the real decision cheap and data-driven.\nNote: options differ in kind (persisted cache vs in-process memo vs measure first) \u2014 no completeness score.\nPros / cons:\nA) Persist the graph with the job on attempt 1\n \u2705 Retries never recompute; cost is paid once per job regardless of which worker instance retries (human: ~1 day / CC: ~20 min)\n \u2705 Simple to reason about once the invalidation rule (payload version) is in place\n \u274c Adds a write and stored blob to every job, including the ones that never retry; stale-graph bugs if the payload changes between attempts\nB) In-process memo (bounded LRU)\n \u2705 No persistence, no schema change; a few lines around the graph builder (human: ~2 hr / CC: ~10 min)\n \u2705 Zero cost on the happy path beyond a map insert\n \u274c Retries after a 10-minute delay usually land on a different worker instance, so the hit rate is low and unpredictable\nC) Measure first: timing metrics + p95 budget (recommended)\n \u2705 Two metrics (graph compute ms, payload bytes) per attempt tell you whether this is 2 ms or 2 s before anyone writes cache code (human: ~1 hr / CC: ~5 min)\n \u2705 No happy-path cost, no staleness risk, and the retry-policy module already logs per attempt (D7) so the hook point exists\n \u274c If the rebuild is genuinely slow, retries stay expensive until the follow-up lands\nNet: Add a write to every job to save CPU on the few that retry, or spend an hour on metrics and decide with numbers.": "Persist the graph with the job" + }, + "unansweredQuestionIndices": [], + "answeredAt": "2026-09-29T20:04:03.405Z" + }, + { + "sessionId": "0117efe3-a002-42c1-aa59-d45032643e13", + "toolUseId": "toolu_01WjGqEAsM8pPCNSzxpWyPSV", + "questions": [ + { + "question": "D10 \u2014 Track dead-letter retention as a TODO, skip it, or build it now?\nProject/branch/task: main branch, retry-framework plan; follow-up to the dead-letter store approved in D3.\nELI10: The dead-letter store keeps every job that ran out of retries or hit a fatal error. Nothing ever removes them. That is fine for months, then the table is large, the growth alert query slows, and nobody remembers why. A purge job with a retention period fixes it, but the retention period is a judgment call about how long failed-job evidence must stay around.\nStakes if we pick wrong: build it now with the wrong retention and you delete evidence of lost work; skip it and the store becomes an unbounded table someone discovers during an incident.\nRecommendation: A because the store is new, growth is slow, and the retention period deserves an owner's answer rather than a default picked inside a retry PR; the TODO carries a concrete trigger.\nCompleteness: A=6/10, B=2/10, C=10/10\nPros / cons:\nA) Add to TODOS.md with a trigger (recommended)\n \u2705 Keeps this PR right-sized: the retry framework ships without a retention debate attached\n \u2705 Trigger (10k rows or 3 months) means the TODO fires before growth matters (human: ~5 min / CC: ~1 min now)\n \u274c Unbounded growth until someone acts on the TODO; the ceiling is a slow query, not data loss\nB) Skip\n \u2705 Nothing to track or build\n \u2705 Zero effort now\n \u274c The store grows forever with no record that anyone considered it\nC) Build now in this PR\n \u2705 Store ships bounded from day one: purge job, 90-day default, keep flag, tests (human: ~2 hr / CC: ~10 min)\n \u2705 No follow-up to forget\n \u274c Expands this PR with a scheduled job and a retention default nobody has agreed to; deletes evidence if the default is wrong\nNet: A tracked follow-up with a trigger, versus a bigger PR that guesses how long failed-job evidence should live.", + "header": "DLQ retention", + "multiSelect": false, + "options": [ + { + "label": "Add to TODOS.md (recommended)", + "description": "Record the TODO (what/why/pros/cons/context/depends-on) with trigger: build when the dead-letter store passes 10k rows or at 3 months, whichever first. Human ~5 min / CC ~1 min. Completeness 6/10." + }, + { + "label": "Skip \u2014 not valuable enough", + "description": "Do not track retention. Completeness 2/10." + }, + { + "label": "Build it now in this PR", + "description": "Scheduled purge job, retention config default 90 days, keep flag, tests, shipped with the dead-letter store. Human ~2 hr / CC ~10 min. Completeness 10/10." + } + ] + } + ], + "answered": true, + "failed": false, + "answers": { + "D10 \u2014 Track dead-letter retention as a TODO, skip it, or build it now?\nProject/branch/task: main branch, retry-framework plan; follow-up to the dead-letter store approved in D3.\nELI10: The dead-letter store keeps every job that ran out of retries or hit a fatal error. Nothing ever removes them. That is fine for months, then the table is large, the growth alert query slows, and nobody remembers why. A purge job with a retention period fixes it, but the retention period is a judgment call about how long failed-job evidence must stay around.\nStakes if we pick wrong: build it now with the wrong retention and you delete evidence of lost work; skip it and the store becomes an unbounded table someone discovers during an incident.\nRecommendation: A because the store is new, growth is slow, and the retention period deserves an owner's answer rather than a default picked inside a retry PR; the TODO carries a concrete trigger.\nCompleteness: A=6/10, B=2/10, C=10/10\nPros / cons:\nA) Add to TODOS.md with a trigger (recommended)\n \u2705 Keeps this PR right-sized: the retry framework ships without a retention debate attached\n \u2705 Trigger (10k rows or 3 months) means the TODO fires before growth matters (human: ~5 min / CC: ~1 min now)\n \u274c Unbounded growth until someone acts on the TODO; the ceiling is a slow query, not data loss\nB) Skip\n \u2705 Nothing to track or build\n \u2705 Zero effort now\n \u274c The store grows forever with no record that anyone considered it\nC) Build now in this PR\n \u2705 Store ships bounded from day one: purge job, 90-day default, keep flag, tests (human: ~2 hr / CC: ~10 min)\n \u2705 No follow-up to forget\n \u274c Expands this PR with a scheduled job and a retention default nobody has agreed to; deletes evidence if the default is wrong\nNet: A tracked follow-up with a trigger, versus a bigger PR that guesses how long failed-job evidence should live.": "Add to TODOS.md (recommended)" + }, + "unansweredQuestionIndices": [], + "answeredAt": "2026-09-29T20:05:54.778Z" + }, + { + "sessionId": "0117efe3-a002-42c1-aa59-d45032643e13", + "toolUseId": "toolu_01QrMt8ZhcL4wV3cKEjo3MsA", + "questions": [ + { + "question": "D11 \u2014 Track the webhook event-id header and at-least-once opt-in as a TODO, skip it, or ship the header now?\nProject/branch/task: main branch, retry-framework plan; follow-up to D2 (webhook worker stays at-most-once).\nELI10: D2 kept the promise that a webhook is never sent twice, which means a send that times out is still lost. The usual fix is to stamp every event with a stable id so receivers can ignore duplicates, and then retry freely. That is a contract change, so it was declined for this PR. The question is whether to track it, drop it, or at least ship the harmless id header now so receivers can start deduping before the semantics ever change.\nStakes if we pick wrong: lost webhook events keep landing in dead-letter with no plan to stop the loss; or a header change rides along in a retry PR without receiver communication.\nRecommendation: A because this is a receiver-facing contract change that deserves its own PR and docs, and the dead-letter store (D3) will produce the loss numbers that justify it; the trigger is concrete.\nCompleteness: A=6/10, B=2/10, C=8/10\nPros / cons:\nA) Add to TODOS.md with a trigger (recommended)\n \u2705 Keeps the retry PR free of webhook contract changes; the TODO fires on measured loss (human: ~5 min / CC: ~1 min now)\n \u2705 Dead-letter counts of post-send failures give the case for it with real numbers\n \u274c Webhook events lost to timeouts stay lost until the TODO is acted on\nB) Skip\n \u2705 Nothing to track\n \u2705 Zero effort now\n \u274c No record that at-most-once was a deliberate trade with a known cost\nC) Ship the stable event-id header now, at-least-once later\n \u2705 Receivers can start deduping today; the header is harmless under at-most-once (human: ~2 hr / CC: ~10 min)\n \u2705 Makes the eventual semantics change a config flip instead of a payload change\n \u274c Adds a webhook payload change and receiver docs to a retry PR; still needs the TODO for the semantics\nNet: Track it with a loss-based trigger, or ship a small header change now inside a PR about retries.", + "header": "Webhook TODO", + "multiSelect": false, + "options": [ + { + "label": "Add to TODOS.md (recommended)", + "description": "Record the TODO with trigger: post-send dead-letter entries exceed 1% of webhook sends in any week, or a receiver requests redelivery. Human ~5 min / CC ~1 min. Completeness 6/10." + }, + { + "label": "Skip \u2014 not valuable enough", + "description": "Do not track. Completeness 2/10." + }, + { + "label": "Ship the header now", + "description": "Add a stable per-event id header to every webhook delivery in this PR; D2 semantics unchanged; at-least-once opt-in remains a TODO. Human ~2 hr / CC ~10 min. Completeness 8/10." + } + ] + } + ], + "answered": true, + "failed": false, + "answers": { + "D11 \u2014 Track the webhook event-id header and at-least-once opt-in as a TODO, skip it, or ship the header now?\nProject/branch/task: main branch, retry-framework plan; follow-up to D2 (webhook worker stays at-most-once).\nELI10: D2 kept the promise that a webhook is never sent twice, which means a send that times out is still lost. The usual fix is to stamp every event with a stable id so receivers can ignore duplicates, and then retry freely. That is a contract change, so it was declined for this PR. The question is whether to track it, drop it, or at least ship the harmless id header now so receivers can start deduping before the semantics ever change.\nStakes if we pick wrong: lost webhook events keep landing in dead-letter with no plan to stop the loss; or a header change rides along in a retry PR without receiver communication.\nRecommendation: A because this is a receiver-facing contract change that deserves its own PR and docs, and the dead-letter store (D3) will produce the loss numbers that justify it; the trigger is concrete.\nCompleteness: A=6/10, B=2/10, C=8/10\nPros / cons:\nA) Add to TODOS.md with a trigger (recommended)\n \u2705 Keeps the retry PR free of webhook contract changes; the TODO fires on measured loss (human: ~5 min / CC: ~1 min now)\n \u2705 Dead-letter counts of post-send failures give the case for it with real numbers\n \u274c Webhook events lost to timeouts stay lost until the TODO is acted on\nB) Skip\n \u2705 Nothing to track\n \u2705 Zero effort now\n \u274c No record that at-most-once was a deliberate trade with a known cost\nC) Ship the stable event-id header now, at-least-once later\n \u2705 Receivers can start deduping today; the header is harmless under at-most-once (human: ~2 hr / CC: ~10 min)\n \u2705 Makes the eventual semantics change a config flip instead of a payload change\n \u274c Adds a webhook payload change and receiver docs to a retry PR; still needs the TODO for the semantics\nNet: Track it with a loss-based trigger, or ship a small header change now inside a PR about retries.": "Add to TODOS.md (recommended)" + }, + "unansweredQuestionIndices": [], + "answeredAt": "2026-09-29T20:07:04.232Z" + } + ], + "plan": "# Eng Review \u2014 PLAN.md: Add background job retry framework\n\n- **Review target (fixed):** `/tmp/gstack-plan-count-vrYrwf/PLAN.md` \u2014 \"Plan: Add background job retry framework\"\n- **Report file:** `/tmp/gstack-e2e-plan-eng-batching-PLhMrg/gstack-test-plan-eng-batching.md` (destination explicitly requested by the user)\n- **Skill:** `/plan-eng-review` \u00b7 session `256191-1790711293-1222b505` \u00b7 2026-09-29 \u00b7 branch `main` @ `82eaa12`\n- **Evidence available:** the repository contains only `PLAN.md` and `CLAUDE.md`. No worker files, job library, or `processWebhookJob()` source exist in this checkout. Findings below quote the plan text (file:line) and are calibrated as plan-level, not code-verified.\n\n## Original plan (unchanged copy of PLAN.md lines 4-24)\n\n```markdown\n# Plan: Add background job retry framework\n\n## Architecture\nWe'll roll a custom exponential-backoff scheduler inline in each worker\nrather than use the existing job library's built-in retry hooks. Same\nshape as the library version, but we want full control over the curve.\n\n## Code quality\nThe retry envelope (compute delay, log attempt, dispatch) is duplicated\nacross 5 worker files with copy-pasted bodies. We will leave the\nduplication for now and refactor \"later.\"\n\n## Tests\nThe existing `processWebhookJob()` flow gets rewritten as part of this\nchange. No regression test for the prior at-most-once delivery guarantee\nis planned.\n\n## Performance\nOn every retry we re-fetch the full job payload from the database, then\niterate the payload to recompute the dependency graph. Could cache the\ngraph on the first attempt; not planned.\n```\n\n## Scope Challenge\n\n### A. Assessment\n- **Already solves it:** the job library's built-in retry hooks (PLAN.md:8 admits \"same shape as the library version\"). Library source not in checkout; hook API unverified.\n- **Complexity count (estimate from plan text):** 5 worker files (PLAN.md:13) + `processWebhookJob()` (PLAN.md:17, likely one of the five) = 5\u20136 changed files; 0 new classes/services (scheduler is inline). Under thresholds \u2192 complexity gate B skipped.\n- **Search check:** [Layer 1] library retry hooks + backoff callback; jitter, delay cap, dead-letter, idempotency key are standard practice (AWS Builders' Library; Hookdeck/Svix idempotency guides).\n- **TODOS.md:** none. **Distribution:** no new artifacts.\n\n### C. Findings (plan-level; no code in checkout)\n1. `[P1] (confidence: 8/10) PLAN.md:7-9` \u2014 rebuilding a retry scheduler the job library already provides. \u2192 R1 / D1\n2. `[P1] (confidence: 7/10) PLAN.md:17-19` \u2014 retrying `processWebhookJob()` changes at-most-once to at-least-once delivery; semantics change, not just a missing test. \u2192 Section 1\n3. `[P2] (confidence: 7/10) PLAN.md:7-9` \u2014 retry policy bounds unspecified (max attempts, delay cap, jitter, dead-letter, retryable vs fatal errors). \u2192 Section 1\n4. `[P2] (confidence: 7/10) PLAN.md:12-14` \u2014 five copy-pasted retry envelopes. \u2192 Section 2\n5. `[P1] (confidence: 8/10) PLAN.md:18-19` \u2014 no regression test for a rewritten flow with a stated guarantee (Regression Rule). \u2192 Section 3\n6. `[P2] (confidence: 6/10) PLAN.md:22-24` \u2014 full payload refetch + graph recompute on every retry. \u2192 Section 4\n\nScope Challenge result: **scope accepted as-is** (D1 changed the mechanism to library retry hooks; no feature was cut, so this is not a scope reduction). Dispositions: finding 1 accepted via D1 (R1 approved); findings 2\u20136 pending in their sections.\n\n## Section 1 \u2014 Architecture review\n\nWorking plan after D1: all 5 workers retry through the job library's hooks; each worker supplies its own backoff curve.\n\n```\nRETRY STATE MACHINE (per job, owned by the library after D1)\n\n enqueue \u2500\u2500\u25b6 [attempt n] \u2500\u2500success\u2500\u2500\u25b6 DONE\n \u2502\n \u251c\u2500 fatal error (R3d: non-retryable class) \u2500\u2500\u25b6 FAILED \u2500\u2500\u25b6 dead-letter (R3a)\n \u2502\n \u2514\u2500 transient error / timeout\n \u2502\n \u251c\u2500 n >= maxAttempts (R3a) \u2500\u2500\u25b6 FAILED \u2500\u2500\u25b6 dead-letter (R3a)\n \u2502\n \u2514\u2500 delay = min(base\u00b72^n (+ jitter R3b), cap R3c) \u2500\u2500\u25b6 [attempt n+1]\n\n Webhook worker only: timeout after the request was written is AMBIGUOUS \u2014\n the receiver may already have the event. A retry here = possible duplicate (R2).\n```\n\nFindings:\n- `[P1] (confidence: 7/10) PLAN.md:17-19` \u2014 \"The existing `processWebhookJob()` flow gets rewritten ... prior at-most-once delivery guarantee.\" Adding retries flips the webhook worker from at-most-once to at-least-once: a retry after an ambiguous timeout can deliver the same event twice. The plan treats this as a missing test; it is a delivery-contract change for every receiver. \u2192 R2 / D2\n- `[P2] (confidence: 7/10) PLAN.md:7-9` \u2014 \"custom exponential-backoff scheduler ... full control over the curve.\" The curve is named but no bound is: no max attempts or terminal handling (dead-letter), no jitter, no delay cap, no retryable-vs-fatal error classification. Each is an independent choice. \u2192 R3a (terminal handling), R3b (jitter), R3c (delay cap), R3d (error classification)\n- `[P2] (confidence: 6/10) PLAN.md:22` \u2014 \"On every retry we re-fetch the full job payload from the database\" implies attempt state lives in the DB between attempts. With D1 (library hooks) attempt counting and persistence are library-owned; no separate decision. Performance side \u2192 Section 4 (R6).\n- Realistic production failure: the downstream webhook receiver is down for 2 hours. All webhook jobs fail together, then retry together when it recovers (thundering herd without jitter, R3b), and jobs past max attempts must land somewhere visible (R3a) rather than vanish.\n- Diagram: the retry state machine above belongs inline in the shared retry config module once written.\n- Distribution: no new artifacts; no CI/CD change.\n\nSection 1 dispositions: finding 1 \u2192 R2 approved A (D2, at-most-once kept); finding 2 \u2192 R3a approved A (D3), R3b approved A (D4), R3c approved A (D5), R3d approved A (D6); finding 3 \u2192 resolved by D1 (library owns attempt state), performance side in Section 4. Section 1 total: 3 findings, 0 open.\n\n## Section 2 \u2014 Code quality review\n\nFindings:\n- `[P2] (confidence: 7/10) PLAN.md:12-14` \u2014 \"duplicated across 5 worker files with copy-pasted bodies. We will leave the duplication for now and refactor 'later.'\" After D1\u2013D6 the per-worker glue is identical by construction; five copies of jitter/cap/classifier/dead-letter code is the drift risk, not a style nit. Shared-code rubric evidence is in the R4 record. \u2192 R4 / D7\n- `[P2] (confidence: 6/10) PLAN.md:7-9` \u2014 error-handling gap: the plan never says what happens when the dead-letter write itself fails (store down). Necessary implementation of the approved D3 contract, not a new choice: if `toDeadLetter` throws, the job stays in the library's failed state, an error log with the original error and the dead-letter failure fires, and a metric increments. Never swallow both errors. Test required (Section 3).\n- `[P2] (confidence: 6/10) PLAN.md:22` \u2014 edge case: the job row is deleted between attempts, so the refetch returns nothing. Under D6 \"missing record\" is fatal \u2192 dead-letter immediately with the payload reference. Add to each worker's fatal list explicitly; test required.\n- `[P3] (confidence: 6/10)` \u2014 config edge cases: `maxAttempts` 0 or negative, `maxDelay` below base, missing per-worker override. Validate at worker startup and fail fast (part of R4 option A/C; in B, five copies of the validation). Test required.\n- Diagrams: the retry state machine (Section 1) belongs inline as a comment in the shared module (R4 A/C) or in each worker (R4 B).\n- Debt check: with D1\u2013D6 approved, no premature abstraction remains in the plan; the only fragility is the duplication itself.\n\nSection 2 dispositions: finding 1 \u2192 R4 approved A (D7, shared module); findings 2\u20134 carried as necessary implementation/proof of D3, D6 and D7 (no new choice). Section 2 total: 4 findings, 0 open.\n\n## Section 3 \u2014 Test review\n\n**Test framework detection:** CLAUDE.md has no Testing section. Auto-detect in the checkout: no runtime markers, no test config, `TESTFILES:0`. Framework unknown; the plan proposes none, so no framework question. Assertions below are framework-neutral.\n\n**Step 1\u20132: traced codepaths and user flows (all proposed; no runnable source in checkout).** Entry points: each worker's job handler wired to the library retry hooks (D1) via the shared module (D7); the webhook worker's send path (D2); the dead-letter store and its replay path (D3).\n\n**Step 3\u20134: coverage diagram**\n\n```\nCODE PATHS (proposed) USER / OPERATOR FLOWS\n[+] retry-policy module (D7) [+] Dead-letter operations (D3)\n \u251c\u2500\u2500 buildBackoff() \u251c\u2500\u2500 [GAP] [\u2192E2E] alert fires when dead-letter count grows\n \u2502 \u251c\u2500\u2500 [GAP] 0 \u2264 delay \u2264 base\u00b7mult^n (seeded RNG, D4) \u251c\u2500\u2500 [GAP] operator sees last error + attempt history\n \u2502 \u251c\u2500\u2500 [GAP] delay capped at maxDelay for large n (D5) \u251c\u2500\u2500 [GAP] [\u2192E2E] replay re-enqueues a non-webhook job once\n \u2502 \u2514\u2500\u2500 [GAP] defaults apply when worker sets nothing \u2514\u2500\u2500 [GAP] webhook replay shows \"may duplicate\" warning\n \u251c\u2500\u2500 classify()\n \u2502 \u251c\u2500\u2500 [GAP] retryable class \u2192 retryable [+] Webhook receiver experience (D2)\n \u2502 \u251c\u2500\u2500 [GAP] fatal class \u2192 fatal \u251c\u2500\u2500 [GAP] [\u2192E2E] receiver gets each event exactly once\n \u2502 \u2514\u2500\u2500 [GAP] unknown error \u2192 retryable (D6 default) \u2514\u2500\u2500 [GAP] receiver down 2h: no duplicate on recovery\n \u251c\u2500\u2500 toDeadLetter()\n \u2502 \u251c\u2500\u2500 [GAP] persists record + emits metric [+] Error states\n \u2502 \u2514\u2500\u2500 [GAP] persist fails \u2192 job stays failed, both errors logged \u251c\u2500\u2500 [GAP] bug deploy: fatal \u2192 dead-letter on attempt 1\n \u251c\u2500\u2500 logAttempt() \u2514\u2500\u2500 [GAP] poisoned job: stops at maxAttempts, parked\n \u2502 \u2514\u2500\u2500 [GAP] fields: job id, attempt, delay, error class\n \u2514\u2500\u2500 validateConfig()\n \u251c\u2500\u2500 [GAP] maxAttempts < 1 \u2192 startup failure\n \u2514\u2500\u2500 [GAP] maxDelay < base \u2192 startup failure\n[+] 4 non-webhook workers \u00d7 hook wiring (D1)\n \u251c\u2500\u2500 [GAP] [\u2192E2E] transient error \u2192 retried with module delay\n \u251c\u2500\u2500 [GAP] [\u2192E2E] fatal error \u2192 dead-letter, no attempts consumed\n \u251c\u2500\u2500 [GAP] [\u2192E2E] exhaustion at maxAttempts \u2192 dead-letter\n \u2514\u2500\u2500 [GAP] job row deleted between attempts \u2192 fatal\n[+] processWebhookJob() (D2) \u2014 REGRESSION, CRITICAL \u2192 R5 / D8\n \u251c\u2500\u2500 [GAP] pre-send failure (refused/DNS/TLS/local) \u2192 retry, then exactly 1 send\n \u251c\u2500\u2500 [GAP] timeout after send \u2192 0 further sends, dead-letter entry\n \u251c\u2500\u2500 [GAP] 5xx after send \u2192 0 further sends, dead-letter entry\n \u251c\u2500\u2500 [GAP] connection reset mid-response \u2192 0 further sends\n \u251c\u2500\u2500 [GAP] success \u2192 1 send, no dead-letter\n \u2514\u2500\u2500 [GAP] [\u2192E2E] attempt count survives worker restart mid-curve\n\nLLM integration: none in this plan \u2014 no eval scope.\n\nCOVERAGE: 0/27 paths tested (0%) | Code paths: 0/20 (0%) | User flows: 0/7 (0%)\nQUALITY: \u2605\u2605\u2605:0 \u2605\u2605:0 \u2605:0 | GAPS: 27 (8 E2E, 0 eval)\n```\n\nLegend: \u2605\u2605\u2605 behavior + edge + error | \u2605\u2605 happy path | \u2605 smoke check | [\u2192E2E] = needs integration test | [\u2192EVAL] = needs LLM eval\n\n**Regression Rule:** the `processWebhookJob()` rewrite puts an existing guarantee at risk with no coverage planned (PLAN.md:18-19). Behavior to preserve (fixed by D2): at most one send per event, ever. Intentional change: pre-send failures now retry. Coverage is required; D8 settles how. \u2192 R5\n\n**Step 5: tests to add.** Required proof of approved behaviors (D3\u2013D7), no new choice: every `[GAP]` under the retry-policy module and the 4 workers above, as unit tests in the module's suite plus one integration test per worker through the real hooks. Pending: the webhook regression contract's assertions and depth (R5 / D8). Test Plan Artifact is written after D8.\n\nSection 3 dispositions: CRITICAL regression gap \u2192 R5 approved A (D8, unit + integration); 26 other gaps carried as required proof of D3\u2013D7 (no new choice). Test Plan Artifact written (path in Completion summary). Section 3 total: 27 gaps identified, 0 open decisions.\n\n## Section 4 \u2014 Performance review\n\nFindings:\n- `[P2] (confidence: 6/10) PLAN.md:22-24` \u2014 \"On every retry we re-fetch the full job payload ... recompute the dependency graph. Could cache the graph on the first attempt; not planned.\" Under D1 the refetch is the library's normal dequeue; only the graph recompute is extra, bounded to maxAttempts (D3) per failing job. Unmeasured. Medium confidence, verify this is actually an issue. \u2192 R6 / D9\n- `[P3] (confidence: 5/10)` \u2014 dead-letter store (D3) grows without bound if entries are never replayed or purged. No retention policy in scope. Medium confidence. \u2192 TODO candidate 1 (retention/purge policy).\n- `[P3] (confidence: 6/10)` \u2014 the dead-letter growth alert (D3) should read a counter metric emitted by `toDeadLetter`, not run `COUNT(*)` on the store per job. Implementation guidance inside the approved D3 work; no new choice.\n- N+1: none introduced; the per-attempt job load is one row by id. Memory: jitter RNG and classifier lists are negligible; payload size unknown (metric proposed in R6 option C).\n\nSection 4 dispositions: finding 1 \u2192 R6 approved A (D9, persist the graph with the job; user's call over the measure-first recommendation, adds a schema migration and 4 tests); finding 2 \u2192 T1 (TODO question D10); finding 3 \u2192 guidance inside approved D3 work. Section 4 total: 3 findings, 0 open.\n\n## Decision ledger\n\n### R1: Retry scheduling mechanism (library hooks vs custom inline scheduler)\nFinding: #1, P1, confidence 8/10, PLAN.md:7-9, reviewer: plan-eng-review (Claude)\nPlan baseline: original proposal \u2014 custom exponential-backoff scheduler inline in each of 5 workers; library retry hooks bypassed (PLAN.md:7-9). Nothing approved yet.\nRuntime evidence: unknown \u2014 job library not in checkout. Plan states the library version has the same shape (PLAN.md:8). Whether its hooks accept a custom curve callback is unverified.\nComparison grid:\n\n| Choice | Current | A) Library hooks + custom curve | B) Custom inline scheduler | C) Investigate hook API first |\n|---|---|---|---|---|\n| R1 retry scheduling mechanism | custom inline scheduler (proposed, unapproved) | library retry hooks; per-worker backoff curve supplied as a callback/config | custom scheduler inline in each worker, as planned | undecided; bounded ~30 min read of the library's hook signature/docs, then return to R1 |\n| R2 webhook delivery semantics | pending | pending | pending | pending |\n| R3 retry policy bounds (attempts/cap/jitter/dead-letter) | pending | pending (library config likely hosts them; not decided here) | pending | pending |\n| R4 envelope duplication across 5 workers | pending | pending (largely dissolves if library owns delay+dispatch; not decided here) | pending | pending |\n| R5 regression test for at-most-once | pending | pending | pending | pending |\n| R6 payload refetch / graph cache | pending | pending | pending | pending |\n\nQuestion D1:\nD1 \u2014 Reuse the job library's retry hooks or roll a custom scheduler?\nProject/branch/task: main branch, retry-framework plan; adding retries to 5 background workers.\nELI10: The job library you already use has retry hooks built in, and your plan says your custom version would be \"the same shape.\" Building your own copy inside each worker means five hand-written schedulers to keep correct, versus configuring one curve the library already knows how to run. The plan's reason for going custom is \"full control over the curve,\" and most retry hook APIs give you that through a backoff callback.\nStakes if we pick wrong: five bespoke schedulers drift apart, each grows its own bugs (no jitter, no cap, retry storms), and nobody at 3am knows which curve a given worker actually runs.\nRecommendation: A because the plan admits the library version has the same shape, and a custom curve is usually a config callback, not a new scheduler.\nCompleteness: A=9/10, B=5/10, C=n/a (investigation, decides nothing)\nPros / cons:\nA) Library hooks + custom curve (recommended)\n \u2705 One scheduler the library already tests; the curve becomes a per-worker config function (human: ~1 day / CC: ~20 min)\n \u2705 Attempt-count persistence, max attempts and dead-letter usually come along for free with the hooks\n \u274c If the hook API cannot accept an arbitrary curve function, that is a new fact and R1 reopens\nB) Custom inline scheduler (as planned)\n \u2705 Total control of delay math and logging, independent of the library's release cycle\n \u2705 No dependency on hook semantics nobody has read yet (human: ~3 days / CC: ~1 hr)\n \u274c Five hand-rolled schedulers to keep in sync, plus reimplementing attempt persistence and terminal handling\nC) Investigate first: bounded read of the hook API\n \u2705 Settles \"can the hooks take a custom curve\" with evidence before committing either way\n \u2705 Cheap: ~30 min human / ~3 min CC reading the hook signature and docs\n \u274c Decides nothing; R1 stays pending and the review pauses on this question\nNet: You are trading a library dependency you already carry for five copies of scheduler code you would own forever.\nHeader: Retry mechanism\nOptions:\nA) Library hooks + custom curve (recommended)\nUse the job library's built-in retry hooks; supply each worker's backoff curve as a callback/config. One scheduler the library already tests. Attempt persistence, max attempts and dead-letter usually included. Human ~1 day / CC ~20 min. Risk: if the hook API cannot take an arbitrary curve, R1 reopens. Completeness 9/10.\nB) Custom inline scheduler (as planned)\nRoll the exponential-backoff scheduler inline in each of the 5 workers as PLAN.md:7-9 proposes. Total control of delay math and logging. Human ~3 days / CC ~1 hr. Cost: five schedulers to keep in sync, plus attempt persistence and terminal handling rebuilt by hand. Completeness 5/10.\nC) Investigate hook API first\nBounded ~30 min human / ~3 min CC read of the library's retry hook signature and docs, then return to this question. Approves nothing; R1 stays pending; R2\u2013R6 unchanged.\n\nState: approved\nActual answer: A) Library hooks + custom curve \u2014 user answer to D1\nAccepted scope: Use the job library's built-in retry hooks in all 5 workers (including the `processWebhookJob()` worker); supply each worker's backoff curve as a callback/config; do not build the custom inline scheduler from PLAN.md:7-9. Condition carried: if the hook API cannot accept an arbitrary curve function, R1 reopens as a new fact. R2\u2013R6 remain pending and are unchanged by this answer.\nHistory: none\n\n### R2: Delivery guarantee for `processWebhookJob()` once it can retry\nFinding: Section 1 finding 1 (Scope Challenge #2), P1, confidence 7/10, PLAN.md:17-19, reviewer: plan-eng-review (Claude)\nPlan baseline: existing behavior is at-most-once delivery (PLAN.md:18). The proposed rewrite adds retries and states no delivery guarantee. Nothing approved for R2.\nRuntime evidence: unknown \u2014 `processWebhookJob()` source not in checkout. Industry practice (Hookdeck, Svix, Stripe, Shopify) is at-least-once delivery with a stable per-event idempotency key.\nComparison grid:\n\n| Choice | Current | A) Keep at-most-once | B) At-least-once + idempotency key | C) Exclude webhook worker from retries |\n|---|---|---|---|---|\n| R2 webhook delivery semantics | at-most-once (existing); rewrite unspecified | at-most-once kept; retry only failures provably raised before the request was written (connection refused, DNS, local error); timeouts/5xx-after-send still drop | at-least-once; a stable event id/idempotency key header constant across attempts; documented contract change + receiver migration note | webhook worker keeps today's at-most-once behavior with no retries; other 4 workers retry per D1 |\n| R1 retry mechanism | approved A (D1) | fixed | fixed | fixed for the 4 non-webhook workers; webhook worker untouched |\n| R3a\u2013R3d retry bounds | pending | pending | pending | pending |\n| R4 envelope duplication | pending | pending | pending | pending |\n| R5 regression test for the webhook flow | pending (behavior to preserve depends on R2) | pending | pending | pending |\n| R6 payload refetch / graph cache | pending | pending | pending | pending |\n\nQuestion D2:\nD2 \u2014 What delivery guarantee does processWebhookJob() keep once it can retry?\nProject/branch/task: main branch, retry-framework plan; the webhook worker is one of the 5 workers gaining retries through library hooks (D1).\nELI10: Today the webhook worker sends each event at most once: if the send fails or times out, the event is dropped, never duplicated. A retry cannot tell \"the request never arrived\" apart from \"it arrived but the response got lost,\" so any retry after a timeout can deliver the same event twice. Adding retries silently flips the guarantee from at-most-once to at-least-once. That is a contract change your webhook receivers depend on, and the plan does not name it.\nStakes if we pick wrong: receivers that are not idempotent process duplicate events (double emails, double charges, double state transitions); or, if we keep dropping on ambiguity, the retry framework never fixes the webhook worker's lost events.\nRecommendation: B because losing events is usually worse than duplicates, and a stable idempotency key makes duplicates safe for receivers; this is still a contract call you know better than the review does.\nNote: options differ in kind, not coverage \u2014 no completeness score.\nPros / cons:\nA) Keep at-most-once: retry only provably-unsent failures\n \u2705 No duplicate deliveries ever; existing receivers keep working with no change on their side\n \u2705 Still recovers the clear cases: connection refused, DNS failure, local enqueue error (human: ~1 day / CC: ~30 min)\n \u274c Timeouts and 5xx-after-send still drop events, so the biggest source of loss stays; needs per-attempt failure classification\nB) Move to at-least-once with a stable idempotency key (recommended)\n \u2705 Every event eventually reaches the receiver; retries after timeouts are safe because the event id stays constant across attempts\n \u2705 Matches Stripe, Shopify and Svix practice; receivers dedupe on the key (human: ~1.5 days / CC: ~30 min incl. docs)\n \u274c Contract change: receivers must dedupe; needs a documented header, changelog entry and migration note for existing receivers\nC) Exclude processWebhookJob() from retries\n \u2705 Zero semantic change for receivers; the other 4 workers still get retries\n \u2705 Smallest diff and no receiver communication (human: ~1 hr / CC: ~5 min)\n \u274c The webhook worker keeps losing events on every transient failure, which is likely why the plan touched it\nNet: Never-duplicate-but-lossy, never-lossy-but-receivers-must-dedupe, or leave the webhook worker exactly as it is.\nHeader: Webhook delivery\nOptions:\nA) Keep at-most-once (retry only pre-send failures)\nWebhook worker retries only failures provably raised before the request was written (connection refused, DNS, local error). Timeouts and 5xx-after-send still drop the event. No duplicates; receivers unchanged. Needs per-attempt failure classification. Human ~1 day / CC ~30 min.\nB) At-least-once + idempotency key (recommended)\nWebhook worker retries all transient failures; every delivery carries a stable event id / idempotency key header constant across attempts. Documented contract change with changelog and receiver migration note. Receivers dedupe on the key. Human ~1.5 days / CC ~30 min.\nC) Exclude webhook worker from retries\n`processWebhookJob()` keeps today's at-most-once, no-retry behavior; the other 4 workers retry via library hooks per D1. Smallest diff, no receiver impact, webhook events still lost on transient failure. Human ~1 hr / CC ~5 min.\n\nState: approved\nActual answer: A) Keep at-most-once (retry only pre-send failures) \u2014 user answer to D2 (not the recommended option; user's contract call)\nAccepted scope: `processWebhookJob()` keeps the at-most-once delivery guarantee. It retries (via library hooks, D1) only failures provably raised before any request bytes were written: connection refused, DNS failure, TLS handshake failure, local serialization/enqueue error. Any failure after the request is written (timeout, 5xx, connection reset mid-response) is terminal for that event: no retry, dropped as today, logged with the failure class. Necessary implementation carried as common work: per-attempt pre-send vs post-send failure classification inside the webhook worker, plus its tests (Section 3, R5 regression contract: at most one send per event, ever). No idempotency header, no receiver contract change. R3a\u2013R3d, R4, R6 unchanged and pending; R3d (retryable vs fatal classification for the other 4 workers) remains its own choice.\nHistory: none\n\n### R3a: Terminal handling \u2014 max attempts and what happens when they run out\nFinding: Section 1 finding 2 (Scope Challenge #3), P2, confidence 7/10, PLAN.md:7-9, reviewer: plan-eng-review (Claude)\nPlan baseline: original proposal names \"full control over the curve\" (PLAN.md:9) and no attempt limit or terminal outcome. Nothing approved for R3a.\nRuntime evidence: unknown \u2014 no worker or library source in checkout. Practice: bounded attempts with a dead-letter store; never silently drop (AWS Builders' Library; retry-pattern references).\nComparison grid:\n\n| Choice | Current | A) Bounded + dead-letter + alert | B) Bounded, log and drop | C) Library defaults, unspecified |\n|---|---|---|---|---|\n| R3a terminal handling | unspecified | `maxAttempts` default 5, per-worker override; on exhaustion or fatal error the job lands in a dead-letter store (table or queue) with last error, attempt history and payload ref; metric + alert on dead-letter growth; manual replay path | `maxAttempts` default 5, per-worker override; on exhaustion log at error level with last error and drop the job | whatever the library does by default; not written into the plan |\n| R1 retry mechanism | approved A (D1) | fixed | fixed | fixed |\n| R2 webhook delivery | approved A (D2) | fixed | fixed | fixed |\n| R3b jitter | pending | pending | pending | pending |\n| R3c delay cap | pending | pending | pending | pending |\n| R3d error classification (4 non-webhook workers) | pending | pending | pending | pending |\n| R4 envelope duplication | pending | pending | pending | pending |\n| R5 regression test | pending | pending | pending | pending |\n| R6 payload refetch / graph cache | pending | pending | pending | pending |\n\nQuestion D3:\nD3 \u2014 When a job runs out of retries, where does it go?\nProject/branch/task: main branch, retry-framework plan; retry bounds for all 5 workers running through library hooks (D1).\nELI10: Right now the plan describes the curve between retries but never says how many retries there are or what happens to a job that keeps failing. Without a limit, a poisoned job retries forever and eats worker capacity. With a limit but no landing spot, the job disappears with one log line nobody reads. A dead-letter store keeps the failed job, its payload reference and its last error so someone can inspect and replay it.\nStakes if we pick wrong: either an infinite-retry job starves the queue, or real work silently vanishes after the last attempt and the first sign is a customer asking where their data went.\nRecommendation: A because a dead-letter store plus an alert is a few dozen lines with library hooks, and it turns \"job vanished\" into \"job parked, here is why.\"\nCompleteness: A=10/10, B=5/10, C=3/10\nPros / cons:\nA) Bounded attempts + dead-letter store + alert (recommended)\n \u2705 Exhausted or fatal jobs are kept with last error and attempt history; operators can inspect and replay (human: ~1 day / CC: ~20 min)\n \u2705 Metric and alert on dead-letter growth turns a silent failure into a page at the right time\n \u274c One more table or queue to own, plus a small replay path to build and test\nB) Bounded attempts, log and drop\n \u2705 Simplest bound: `maxAttempts` default 5 per worker, one error log on exhaustion (human: ~2 hr / CC: ~5 min)\n \u2705 No new storage; nothing to operate\n \u274c Exhausted jobs are gone; recovery means replaying from upstream sources by hand, if that is even possible\nC) Leave to library defaults\n \u2705 Zero plan text and zero decision now\n \u2705 Whatever the library does is at least consistent across the 5 workers\n \u274c Nobody knows the limit or the terminal behavior until an incident teaches them; 3am failure mode\nNet: You are trading one small dead-letter store for never having to ask \"where did that job go.\"\nHeader: Retry exhaustion\nOptions:\nA) Bounded + dead-letter + alert (recommended)\n`maxAttempts` default 5 with per-worker override. On exhaustion or fatal error the job lands in a dead-letter store (table or queue) with last error, attempt history and payload reference. Metric and alert on dead-letter growth. Manual replay path. Human ~1 day / CC ~20 min. Completeness 10/10.\nB) Bounded, log and drop\n`maxAttempts` default 5 with per-worker override. On exhaustion, log at error level with the last error and drop the job. No new storage, no replay. Human ~2 hr / CC ~5 min. Completeness 5/10.\nC) Library defaults, unspecified\nDo not write attempt limits or terminal behavior into the plan; accept whatever the library does by default. Completeness 3/10.\n\nState: approved\nActual answer: A) Bounded + dead-letter + alert \u2014 user answer to D3\nAccepted scope: All 5 workers: `maxAttempts` default 5 with per-worker override (library config, D1). On exhaustion or on a fatal (non-retryable) error the job lands in a dead-letter store (table or queue) recording last error, attempt history and payload reference. Metric and alert on dead-letter growth. Manual replay path. Tests for the exhaustion path, the fatal-error path and replay are common work of this behavior. Reconciliation with D2: a webhook post-send failure is terminal for that event and is recorded in the dead-letter store (not auto-retried, no duplicate send); manual replay of a webhook entry is an explicit operator action and the replay UI/docs must say it may duplicate. R3b, R3c, R3d, R4, R5, R6 unchanged and pending.\nHistory: none\n\n### R3b: Jitter on retry delays\nFinding: Section 1 finding 2 (Scope Challenge #3), P2, confidence 7/10, PLAN.md:7-9, reviewer: plan-eng-review (Claude)\nPlan baseline: \"custom exponential-backoff ... full control over the curve\" (PLAN.md:7-9); no jitter mentioned. Nothing approved for R3b.\nRuntime evidence: unknown \u2014 no worker source in checkout. AWS Builders' Library analysis: full jitter gives the least contention and total work; deterministic exponential curves synchronize retries after a shared outage.\nComparison grid:\n\n| Choice | Current | A) Full jitter | B) Equal jitter | C) No jitter |\n|---|---|---|---|---|\n| R3b jitter | unspecified (deterministic curve implied) | delay = random(0, exponentialDelay) inside each worker's curve callback; RNG injectable for tests | delay = exponentialDelay/2 + random(0, exponentialDelay/2); RNG injectable for tests | deterministic exponentialDelay; no randomization |\n| R1 retry mechanism | approved A (D1) | fixed | fixed | fixed |\n| R2 webhook delivery | approved A (D2) | fixed | fixed | fixed |\n| R3a terminal handling | approved A (D3) | fixed | fixed | fixed |\n| R3c delay cap | pending | pending | pending | pending |\n| R3d error classification (4 non-webhook workers) | pending | pending | pending | pending |\n| R4 envelope duplication | pending | pending | pending | pending |\n| R5 regression test | pending | pending | pending | pending |\n| R6 payload refetch / graph cache | pending | pending | pending | pending |\n\nQuestion D4:\nD4 \u2014 Should retry delays be randomized (jitter)?\nProject/branch/task: main branch, retry-framework plan; the backoff curve each worker supplies to the library hooks (D1).\nELI10: When many jobs fail at the same moment because a shared dependency went down, a pure exponential curve makes them all retry at the same moments too, so the recovering dependency gets hit by a wave on every step. Jitter randomizes each job's delay so the retries spread out. It is one line inside the curve callback each worker already supplies.\nStakes if we pick wrong: synchronized retry waves knock a recovering dependency back over (the classic thundering herd); or, with jitter, per-job retry timing becomes slightly less predictable and tests need a seeded random source.\nRecommendation: A because full jitter gives the least contention in AWS's published analysis and costs one line in a callback you are already writing.\nNote: options differ in kind, not coverage \u2014 no completeness score.\nPros / cons:\nA) Full jitter: random(0, exponentialDelay) (recommended)\n \u2705 Best spread of retries and lowest total contention after a shared outage (AWS Builders' Library)\n \u2705 One line inside the D1 curve callback; RNG injected so tests stay deterministic (human: ~1 hr / CC: ~5 min)\n \u274c An individual retry can fire almost immediately; minimum wait is not guaranteed\nB) Equal jitter: half fixed, half random\n \u2705 Guarantees a minimum wait of half the exponential delay while still spreading retries\n \u2705 Same one-line cost and same injectable RNG as full jitter (human: ~1 hr / CC: ~5 min)\n \u274c Slightly more contention than full jitter in the same analysis, and one more parameter to explain\nC) No jitter: deterministic curve\n \u2705 Fully deterministic; trivial to reason about and to assert exact delays in tests\n \u2705 Zero extra code beyond the exponential curve\n \u274c Every job that failed together retries together; retry storms on recovery are the expected outcome\nNet: One random() call now versus a synchronized retry wave the first time a dependency has a bad hour.\nHeader: Jitter\nOptions:\nA) Full jitter (recommended)\ndelay = random(0, exponentialDelay) inside each worker's curve callback. Best spread, lowest contention. RNG injectable so tests are deterministic. Human ~1 hr / CC ~5 min.\nB) Equal jitter\ndelay = exponentialDelay/2 + random(0, exponentialDelay/2). Guarantees a minimum wait; slightly more contention than full jitter. RNG injectable. Human ~1 hr / CC ~5 min.\nC) No jitter\nDeterministic exponential curve, no randomization. Simplest to test; retries synchronize after a shared outage.\n\nState: approved\nActual answer: A) Full jitter \u2014 user answer to D4\nAccepted scope: Every worker's backoff curve callback (D1) applies full jitter: delay = random(0, exponentialDelay). The random source is injectable so unit tests assert exact delays with a seeded RNG. Tests for the jitter bounds (0 \u2264 delay \u2264 exponentialDelay) are common work of this behavior. R3c, R3d, R4, R5, R6 unchanged and pending.\nHistory: none\n\n### R3c: Ceiling on the backoff delay (delay cap)\nFinding: Section 1 finding 2 (Scope Challenge #3), P2, confidence 7/10, PLAN.md:7-9, reviewer: plan-eng-review (Claude)\nPlan baseline: exponential backoff with \"full control over the curve\" (PLAN.md:7-9); no base delay, multiplier or ceiling stated. Nothing approved for R3c.\nRuntime evidence: unknown \u2014 no worker source in checkout. With D3's per-worker `maxAttempts` override, an uncapped doubling curve from a 1 s base reaches ~4.5 h at attempt 15 and ~6 days at attempt 20.\nComparison grid:\n\n| Choice | Current | A) Cap each delay | B) No cap |\n|---|---|---|---|\n| R3c delay cap | unspecified | delay = min(jitteredExponential, maxDelay); `maxDelay` default 10 min, per-worker override; curve defaults documented as base 1 s, multiplier 2, per-worker override | no ceiling; delay follows the raw exponential curve |\n| R1 retry mechanism | approved A (D1) | fixed | fixed |\n| R2 webhook delivery | approved A (D2) | fixed | fixed |\n| R3a terminal handling | approved A (D3) | fixed | fixed |\n| R3b jitter | approved A (D4) | fixed | fixed |\n| R3d error classification (4 non-webhook workers) | pending | pending | pending |\n| R4 envelope duplication | pending | pending | pending |\n| R5 regression test | pending | pending | pending |\n| R6 payload refetch / graph cache | pending | pending | pending |\n\nQuestion D5:\nD5 \u2014 Should the backoff delay have a ceiling?\nProject/branch/task: main branch, retry-framework plan; the curve parameters each worker passes to the library hooks (D1, jittered per D4).\nELI10: Exponential backoff doubles the wait after each failure. That is fine for 5 attempts, but D3 lets each worker raise its attempt count, and a worker set to 15 attempts from a 1 second base would wait about 4.5 hours before its last try; at 20 attempts it would wait 6 days. A cap says \"never wait longer than X between attempts,\" so the curve grows and then flattens. It is one min() call in the callback.\nStakes if we pick wrong: without a cap, a worker with a higher attempt count silently turns into a multi-day wait that looks like a stuck job; with a cap, one more number to document per worker.\nRecommendation: A because the cap is one min() and it makes \"how long can this job be delayed\" a question with an answer.\nCompleteness: A=9/10, B=4/10\nPros / cons:\nA) Cap each delay: default 10 min, per-worker override (recommended)\n \u2705 Worst-case wait between attempts is bounded and documented for every worker (human: ~1 hr / CC: ~5 min)\n \u2705 Also pins the curve defaults (base 1 s, multiplier 2) so all 5 workers start from the same documented numbers\n \u274c One more config value per worker to document and keep sane alongside maxAttempts\nB) No cap\n \u2705 Zero code; the curve is exactly the exponential the plan describes\n \u2705 Fewer knobs to explain\n \u274c Any worker that raises maxAttempts past ~12 gets hour-to-day waits nobody intended\nNet: One min() now versus a job that looks stuck for six days the first time someone bumps an attempt count.\nHeader: Delay cap\nOptions:\nA) Cap each delay (recommended)\ndelay = min(jitteredExponential, maxDelay). `maxDelay` default 10 minutes with per-worker override. Curve defaults documented: base 1 s, multiplier 2, per-worker override. Human ~1 hr / CC ~5 min. Completeness 9/10.\nB) No cap\nRaw exponential curve with no ceiling. Zero code, fewer knobs; high attempt counts produce hour-to-day waits. Completeness 4/10.\n\nState: approved\nActual answer: A) Cap each delay \u2014 user answer to D5\nAccepted scope: Every worker's curve callback (D1) computes delay = min(jitteredExponential, maxDelay) with `maxDelay` default 10 minutes and per-worker override. Curve defaults documented: base 1 s, multiplier 2, per-worker override. Tests asserting the cap is honored at high attempt numbers and that defaults apply when a worker sets nothing are common work of this behavior. R3d, R4, R5, R6 unchanged and pending.\nHistory: none\n\n### R3d: Retryable vs fatal error classification (4 non-webhook workers)\nFinding: Section 1 finding 2 (Scope Challenge #3), P2, confidence 7/10, PLAN.md:7-9, reviewer: plan-eng-review (Claude)\nPlan baseline: the plan retries on failure with no distinction between transient and permanent errors (PLAN.md:7-9). D2 already fixed the webhook worker's classification (pre-send vs post-send); this row covers the other 4 workers only. Nothing approved for R3d.\nRuntime evidence: unknown \u2014 no worker source in checkout. Practice: retry timeouts, connection errors, 429/503, DB deadlocks/serialization failures; never retry validation errors, 4xx other than 429, missing records, or programming errors (search check, Section A).\nComparison grid:\n\n| Choice | Current | A) Classify; unknown \u2192 retryable | B) Retry everything to maxAttempts | C) Classify; unknown \u2192 fatal |\n|---|---|---|---|---|\n| R3d error classification (4 non-webhook workers) | unspecified (every error retried) | each worker declares retryable error classes (timeout, connection error, 429/503, deadlock/serialization failure) and fatal classes (validation error, 4xx other than 429, missing record, programming error such as TypeError); fatal \u2192 dead-letter immediately (D3) without consuming attempts; unclassified errors default to retryable | no classification; every error consumes an attempt until `maxAttempts`, then dead-letter (D3) | same declared classes as A; unclassified errors default to fatal \u2192 dead-letter immediately |\n| R1 retry mechanism | approved A (D1) | fixed | fixed | fixed |\n| R2 webhook delivery | approved A (D2) | fixed | fixed | fixed |\n| R3a terminal handling | approved A (D3) | fixed | fixed | fixed |\n| R3b jitter | approved A (D4) | fixed | fixed | fixed |\n| R3c delay cap | approved A (D5) | fixed | fixed | fixed |\n| R4 envelope duplication | pending | pending | pending | pending |\n| R5 regression test | pending | pending | pending | pending |\n| R6 payload refetch / graph cache | pending | pending | pending | pending |\n\nQuestion D6:\nD6 \u2014 Which errors should the four non-webhook workers retry, and which go straight to dead-letter?\nProject/branch/task: main branch, retry-framework plan; error handling inside the 4 non-webhook workers' library retry hooks (D1). The webhook worker's rule is already fixed by D2.\nELI10: Not every failure is worth retrying. A timeout or a \"service busy\" reply will likely pass on the next try. A validation error, a missing record or a bug that throws will fail the same way five times in a row, burning worker time and delaying the dead-letter record (D3) by the whole backoff curve. Classifying errors sends the hopeless ones to dead-letter immediately and spends retries only on the ones that can recover. The open question is what to do with an error nobody has classified yet.\nStakes if we pick wrong: either a bug retries five times per job across a whole queue before anyone sees it, or a transient error that nobody thought to list dead-letters real work on its first failure.\nRecommendation: A because classification is a short list per worker, and defaulting unknown errors to retryable never loses work: the worst case is five wasted attempts, not a dropped job.\nCompleteness: A=10/10, B=5/10, C=8/10\nPros / cons:\nA) Classify; unknown errors retry (recommended)\n \u2705 Hopeless errors (validation, 4xx, missing record, TypeError) land in dead-letter on attempt 1 with the real cause visible (human: ~half day / CC: ~15 min)\n \u2705 Unlisted errors still retry, so a forgotten transient class costs attempts, never data\n \u274c Each worker maintains a small error-class list, and a new fatal class retries needlessly until someone adds it\nB) Retry everything until maxAttempts\n \u2705 No lists to maintain; identical behavior in all 4 workers (human: ~0 / CC: ~0)\n \u2705 Impossible to misclassify a transient error as fatal\n \u274c A deploy with a bug retries every affected job 5 times over the full curve before dead-lettering; queue capacity burns and diagnosis is delayed\nC) Classify; unknown errors are fatal\n \u2705 Zero wasted attempts on anything not explicitly known to be transient (human: ~half day / CC: ~15 min)\n \u2705 Dead-letter fills fast, so new error classes surface quickly\n \u274c Any transient error missing from the list dead-letters real work on its first failure, which is the exact loss the retry framework exists to prevent\nNet: A short list per worker plus a safe default, versus either wasted retries on bugs or lost work on unlisted transients.\nHeader: Error classes\nOptions:\nA) Classify; unknown \u2192 retryable (recommended)\nEach of the 4 workers declares retryable classes (timeout, connection error, 429/503, deadlock/serialization failure) and fatal classes (validation error, 4xx other than 429, missing record, programming error). Fatal \u2192 dead-letter immediately without consuming attempts. Unclassified errors retry. Human ~half day / CC ~15 min. Completeness 10/10.\nB) Retry everything to maxAttempts\nNo classification. Every error consumes an attempt until `maxAttempts`, then dead-letter per D3. Zero code; bugs retry 5 times per job. Completeness 5/10.\nC) Classify; unknown \u2192 fatal\nSame declared classes as A, but unclassified errors go to dead-letter immediately. No wasted attempts; unlisted transient errors lose work on first failure. Human ~half day / CC ~15 min. Completeness 8/10.\n\nState: approved\nActual answer: A) Classify; unknown \u2192 retryable \u2014 user answer to D6\nAccepted scope: Each of the 4 non-webhook workers declares retryable error classes (timeout, connection error, 429/503, deadlock/serialization failure) and fatal classes (validation error, 4xx other than 429, missing record, programming error). Fatal errors go to the dead-letter store (D3) immediately without consuming attempts. Unclassified errors are retryable. Tests for each class direction and the unknown-error default are common work of this behavior. Webhook worker classification stays as fixed by D2. R4, R5, R6 unchanged and pending.\nHistory: none\n\n### R4: Shared retry-policy module vs five inline copies\nFinding: Section 2 finding 1 (Scope Challenge #4), P2, confidence 7/10, PLAN.md:12-14, reviewer: plan-eng-review (Claude)\nPlan baseline: \"The retry envelope (compute delay, log attempt, dispatch) is duplicated across 5 worker files with copy-pasted bodies. We will leave the duplication for now and refactor 'later.'\" (PLAN.md:12-14). Nothing approved for R4.\nRuntime evidence: unknown \u2014 worker files not in checkout. After D1 the library owns dispatch and scheduling; what each worker still supplies is identical behavior by construction: the jittered, capped curve callback (D4, D5), the error classifier shape (D6), the dead-letter handoff (D3) and the attempt log line. Callers: the 5 workers named in PLAN.md:13 \u2014 **proposed callers, labelled as plan assumptions, not verified source**.\nShared-code rubric:\n- Callers: 5 proposed workers (PLAN.md:13). Same behavior required by D3\u2013D6 with only per-worker values (attempts, maxDelay, error lists) differing.\n- Reuse before extracting: the library hooks (D1) are the reuse; the helper is the thin glue that configures them identically.\n- Helper contract (small): `buildBackoff({base=1s, multiplier=2, maxDelay=10min, rng})` \u2192 curve callback; `classify(error, {retryable, fatal})` \u2192 `retryable | fatal`; `toDeadLetter(job, error, attempts)` \u2192 persists record + emits metric; `logAttempt(job, n, delay, errorClass)`; config validation at startup (maxAttempts \u2265 1, maxDelay \u2265 base). Blast radius: a helper bug affects all 5 workers, mitigated by the helper's own tests.\n- Line estimate (ranges, plan-level): inline per worker \u2248 25\u201340 lines \u00d7 5 = 125\u2013200 removed; helper \u2248 60\u201380 added; per-worker config \u2248 8 \u00d7 5 = 40 added. Implementation savings \u2248 15\u201390 lines. Tests: one helper suite (~100 lines) replaces five near-identical suites; total change likely still shrinks, but caller-integration tests may make the first PR grow.\nComparison grid:\n\n| Choice | Current | A) One shared retry-policy module | B) Five inline copies (as planned) | C) Extract curve builder only |\n|---|---|---|---|---|\n| R4 envelope duplication | 5 copy-pasted envelopes, refactor \"later\" | one small module (buildBackoff, classify, toDeadLetter, logAttempt, config validation) used by all 5 workers; each worker keeps only its values | each worker carries its own curve, classifier, dead-letter handoff and log line | shared `buildBackoff` only; classifier, dead-letter handoff and log line stay inline in each worker |\n| R1 retry mechanism | approved A (D1) | fixed | fixed | fixed |\n| R2 webhook delivery | approved A (D2) | fixed | fixed | fixed |\n| R3a terminal handling | approved A (D3) | fixed | fixed | fixed |\n| R3b jitter | approved A (D4) | fixed | fixed | fixed |\n| R3c delay cap | approved A (D5) | fixed | fixed | fixed |\n| R3d error classification | approved A (D6) | fixed | fixed | fixed |\n| R5 regression test | pending | pending | pending | pending |\n| R6 payload refetch / graph cache | pending | pending | pending | pending |\n\nQuestion D7:\nD7 \u2014 One shared retry-policy module, or five copies and \"refactor later\"?\nProject/branch/task: main branch, retry-framework plan; how the 5 workers carry the behavior approved in D3\u2013D6.\nELI10: After D1 the library does the scheduling, but every worker still has to hand it the same four things: a jittered, capped curve, an error classifier, a dead-letter handoff and an attempt log line. The plan copies that block into five files and promises to clean up later. \"Later\" for copy-pasted retry code usually means the fifth copy drifts (no cap, wrong jitter) and nobody notices until an incident. The alternative is one small module that each worker configures with its own numbers and error lists.\nStakes if we pick wrong: five curves that silently disagree, five places to fix the next retry bug, and five test suites that each cover a slightly different subset; or, with a shared module, one bug that hits all five workers at once (mitigated by the module's own tests).\nRecommendation: A because the behavior is identical by construction (D3\u2013D6 fixed it), the module is under 100 lines, and it removes more lines than it adds while making the retry rules testable once.\nCompleteness: A=10/10, B=4/10, C=7/10\nPros / cons:\nA) One shared retry-policy module (recommended)\n \u2705 Curve, classifier, dead-letter handoff, attempt log and config validation are tested once and behave the same in all 5 workers (human: ~1 day / CC: ~20 min)\n \u2705 Estimated 15\u201390 implementation lines saved; the helper's test suite replaces five near-duplicate suites\n \u274c A bug in the module reaches all 5 workers; the module's own tests are the guard\nB) Five inline copies, refactor later (as planned)\n \u2705 No shared dependency between workers; each can be changed in isolation (human: ~1.5 days / CC: ~30 min)\n \u2705 Matches the plan text exactly; nothing new to name or place\n \u274c Five copies to keep in sync and five test suites to write; \"later\" rarely arrives for retry glue\nC) Extract the curve builder only\n \u2705 The math most likely to drift (jitter + cap) lives in one place (human: ~1 day / CC: ~15 min)\n \u2705 Smaller shared surface than A\n \u274c Classifier, dead-letter handoff and log line are still copied five times, so most of the duplication and its tests remain\nNet: One under-100-line module now, or five copies plus a promise.\nHeader: Shared module\nOptions:\nA) One shared retry-policy module (recommended)\nSmall module: buildBackoff (base/multiplier/maxDelay/rng), classify (per-worker retryable/fatal lists), toDeadLetter (persist + metric), logAttempt, and startup config validation. All 5 workers use it with their own values. Human ~1 day / CC ~20 min. Completeness 10/10.\nB) Five inline copies (as planned)\nEach worker carries its own curve, classifier, dead-letter handoff and log line; refactor deferred. Human ~1.5 days / CC ~30 min. Completeness 4/10.\nC) Extract curve builder only\nShared buildBackoff (jitter + cap) only; classifier, dead-letter handoff and log line stay inline in each of the 5 workers. Human ~1 day / CC ~15 min. Completeness 7/10.\n\nState: approved\nActual answer: A) One shared retry-policy module \u2014 user answer to D7\nAccepted scope: One small retry-policy module providing `buildBackoff({base, multiplier, maxDelay, rng})`, `classify(error, {retryable, fatal})`, `toDeadLetter(job, error, attempts)` (persist + metric, and on persist failure: job stays in the library's failed state, error log carries both errors, metric increments), `logAttempt(job, n, delay, errorClass)` and startup config validation (maxAttempts \u2265 1, maxDelay \u2265 base). All 5 workers use it with their own values (attempts, maxDelay, error lists); the webhook worker's pre-send/post-send rule (D2) is its classifier input. The retry state machine diagram lives inline in this module. The module's own test suite plus one integration test per worker are common work of this behavior. R5, R6 unchanged and pending.\nHistory: none\n\n### R5: Regression coverage for `processWebhookJob()` at-most-once delivery\nFinding: Section 3 CRITICAL regression gap (Scope Challenge #5), P1, confidence 8/10, PLAN.md:17-19, reviewer: plan-eng-review (Claude)\nPlan baseline: \"No regression test for the prior at-most-once delivery guarantee is planned.\" (PLAN.md:18-19). Behavior to preserve is fixed by D2: at most one HTTP send per event, ever; retries only on pre-send failures; post-send failures terminal \u2192 dead-letter (D3). Intentional differences: pre-send failures now retry (previously dropped). No approved acceptance assertions or test depth yet.\nRuntime evidence: unknown \u2014 `processWebhookJob()` and its tests are not in checkout; TESTFILES:0, no framework detected. Regression Rule: coverage is required; the question is how, not whether.\nComparison grid:\n\n| Choice | Current | A) Unit + integration through the library hooks | B) Unit tests on the classifier and worker only | C) Integration test only |\n|---|---|---|---|---|\n| R5 regression coverage | none planned | unit: fake transport records every send; pre-send failure \u00d7(maxAttempts\u22121) then success \u2192 exactly 1 send; timeout after send \u2192 0 further sends, dead-letter entry; 5xx \u2192 0 further sends, dead-letter; reset mid-response \u2192 0 further sends; success \u2192 1 send, no dead-letter. Integration: real library hooks + fake receiver, same assertions end to end, plus attempt count persisted across a simulated worker restart | the unit assertions from A against the worker with a fake transport; no run through the real library hooks | the integration run from A only; no isolated unit assertions |\n| R1 retry mechanism | approved A (D1) | fixed | fixed | fixed |\n| R2 webhook delivery | approved A (D2) | fixed | fixed | fixed |\n| R3a\u2013R3d | approved A (D3\u2013D6) | fixed | fixed | fixed |\n| R4 shared module | approved A (D7) | fixed | fixed | fixed |\n| R6 payload refetch / graph cache | pending | pending | pending | pending |\n\nQuestion D8:\nD8 \u2014 How do we prove processWebhookJob() still sends each event at most once?\nProject/branch/task: main branch, retry-framework plan; regression coverage for the rewritten webhook worker (D2 fixed the behavior to keep).\nELI10: The webhook worker is being rewritten and it carries a promise to receivers: an event is never sent twice. D2 kept that promise while adding retries for failures that happen before anything is sent. A rewrite with no test for the promise means the first duplicate email or double charge is found by a customer. The test is straightforward: a fake receiver counts sends, and we assert the count is exactly one across every failure pattern. The question is how deep to go: assertions against the worker alone, a run through the real library hooks, or both.\nStakes if we pick wrong: a retry path nobody tested sends duplicates to non-idempotent receivers, or a hook wiring mistake means pre-send failures never actually retry and the framework quietly does nothing for webhooks.\nRecommendation: A because the unit layer pins each failure class cheaply and the integration layer is the only thing that catches hook wiring and attempt persistence, which is where retry bugs actually live.\nCompleteness: A=10/10, B=7/10, C=7/10\nPros / cons:\nA) Unit + integration through the library hooks (recommended)\n \u2705 Every failure class (pre-send, timeout, 5xx, reset, success) asserted in isolation with a recording fake transport (human: ~1 day / CC: ~20 min)\n \u2705 One end-to-end run through the real hooks with a fake receiver catches wiring and attempt-persistence bugs the unit layer cannot see\n \u274c Two test layers to maintain; the integration test needs the library's test harness or an in-process queue\nB) Unit tests only\n \u2705 Fast, deterministic, no queue infrastructure in the test run (human: ~half day / CC: ~10 min)\n \u2705 Pins the classifier and the send-count contract per failure class\n \u274c Never exercises the real hook registration, so a miswired hook passes tests and never retries in production\nC) Integration test only\n \u2705 Exercises the real path receivers depend on (human: ~half day / CC: ~10 min)\n \u2705 Fewer tests to write\n \u274c Slower, and a failure tells you \"something duplicated\" without pointing at which failure class; edge classes get skipped for time\nNet: Cheap isolated assertions plus one real-path run, versus trusting either layer alone to protect a promise made to external receivers.\nHeader: Webhook regression\nOptions:\nA) Unit + integration (recommended)\nUnit: fake transport records every send; assert exactly 1 send after pre-send retries, 0 further sends after timeout/5xx/reset with a dead-letter entry, 1 send on success. Integration: real library hooks + fake receiver, same assertions, plus attempt count survives a simulated worker restart. Human ~1 day / CC ~20 min. Completeness 10/10.\nB) Unit tests only\nThe unit assertions from A against the worker with a fake transport; no run through the real library hooks. Human ~half day / CC ~10 min. Completeness 7/10.\nC) Integration test only\nThe integration run from A only; no isolated per-failure-class assertions. Human ~half day / CC ~10 min. Completeness 7/10.\n\nState: approved\nActual answer: A) Unit + integration \u2014 user answer to D8\nAccepted scope: CRITICAL regression contract for `processWebhookJob()`: behavior preserved = at most one HTTP send per event, ever (D2); intentional change = pre-send failures now retry. Unit tests with a recording fake transport assert: pre-send failure \u00d7(maxAttempts\u22121) then success \u2192 exactly 1 send; timeout after send \u2192 0 further sends and a dead-letter entry; 5xx after send \u2192 0 further sends and a dead-letter entry; connection reset mid-response \u2192 0 further sends; success \u2192 1 send and no dead-letter. Integration test through the real library hooks with a fake receiver repeats those assertions end to end and asserts the attempt count survives a simulated worker restart mid-curve. R6 unchanged and pending.\nHistory: none\n\n### R6: Payload refetch and dependency-graph recompute on every retry\nFinding: Section 4 finding 1 (Scope Challenge #6), P2, confidence 6/10, PLAN.md:22-24, reviewer: plan-eng-review (Claude)\nPlan baseline: \"On every retry we re-fetch the full job payload from the database, then iterate the payload to recompute the dependency graph. Could cache the graph on the first attempt; not planned.\" (PLAN.md:22-24). Nothing approved for R6.\nRuntime evidence: unknown \u2014 no worker source, payload sizes or timings in checkout. Under D1 the \"re-fetch\" is the library dequeuing the job for the attempt, so it is not extra work; the graph recompute is extra CPU, bounded to `maxAttempts` (default 5, D3) per failing job. Failing jobs are the minority; a cache written on the first attempt taxes every successful job to save work on the failing few. Medium confidence, verify with measurements.\nComparison grid:\n\n| Choice | Current | A) Persist the graph with the job | B) In-process memo (LRU by job id + payload hash) | C) Measure first, no cache |\n|---|---|---|---|---|\n| R6 payload refetch / graph cache | recompute on every attempt | compute once on attempt 1, store the graph beside the job row, reuse on retries, invalidate on payload version change | memoize per worker instance, bounded LRU; hits only when the same instance runs the retry | no cache; add per-attempt timing metrics (job load ms, graph compute ms, payload bytes) and a p95 budget; revisit with data |\n| R1\u2013R5 | approved A (D1\u2013D8) | fixed | fixed | fixed |\n\nQuestion D9:\nD9 \u2014 Cache the dependency graph across retries now, or measure first?\nProject/branch/task: main branch, retry-framework plan; per-attempt cost inside the 5 workers running through library hooks (D1).\nELI10: The plan worries that each retry reloads the job and rebuilds its dependency graph from scratch. After D1 the reload is just the library handing the job to the worker, which happens anyway. The rebuild is real extra CPU, but only on retries, and D3 caps those at 5 per failing job. Storing the graph on the first attempt would add a write to every job, including the large majority that succeed first time, to save work on the few that fail. Nobody has measured how long the rebuild takes.\nStakes if we pick wrong: either we add a write and a staleness risk to every job to fix a cost nobody measured, or a genuinely slow rebuild keeps burning worker time on retries and we only find out under load.\nRecommendation: C because an unmeasured optimization that taxes the happy path is the wrong trade; two timing metrics make the real decision cheap and data-driven.\nNote: options differ in kind (persisted cache vs in-process memo vs measure first) \u2014 no completeness score.\nPros / cons:\nA) Persist the graph with the job on attempt 1\n \u2705 Retries never recompute; cost is paid once per job regardless of which worker instance retries (human: ~1 day / CC: ~20 min)\n \u2705 Simple to reason about once the invalidation rule (payload version) is in place\n \u274c Adds a write and stored blob to every job, including the ones that never retry; stale-graph bugs if the payload changes between attempts\nB) In-process memo (bounded LRU)\n \u2705 No persistence, no schema change; a few lines around the graph builder (human: ~2 hr / CC: ~10 min)\n \u2705 Zero cost on the happy path beyond a map insert\n \u274c Retries after a 10-minute delay usually land on a different worker instance, so the hit rate is low and unpredictable\nC) Measure first: timing metrics + p95 budget (recommended)\n \u2705 Two metrics (graph compute ms, payload bytes) per attempt tell you whether this is 2 ms or 2 s before anyone writes cache code (human: ~1 hr / CC: ~5 min)\n \u2705 No happy-path cost, no staleness risk, and the retry-policy module already logs per attempt (D7) so the hook point exists\n \u274c If the rebuild is genuinely slow, retries stay expensive until the follow-up lands\nNet: Add a write to every job to save CPU on the few that retry, or spend an hour on metrics and decide with numbers.\nHeader: Graph cache\nOptions:\nA) Persist the graph with the job\nCompute once on attempt 1, store the graph beside the job row, reuse on retries, invalidate when the payload version changes. Adds a write to every job. Human ~1 day / CC ~20 min.\nB) In-process memo (bounded LRU)\nMemoize the graph per worker instance keyed by job id + payload hash, bounded LRU. No persistence; low hit rate when retries land on another instance. Human ~2 hr / CC ~10 min.\nC) Measure first (recommended)\nNo cache. Add per-attempt timing metrics (job load ms, graph compute ms, payload bytes) via the retry-policy module's attempt log, set a p95 budget, and revisit caching with data. Human ~1 hr / CC ~5 min.\n\nState: approved\nActual answer: A) Persist the graph with the job \u2014 user answer to D9 (not the recommended option; user's call)\nAccepted scope: Compute the dependency graph once on attempt 1 and persist it beside the job row (column or side table keyed by job id) together with the payload version. On retry, load the stored graph when the payload version matches; otherwise recompute and overwrite. A failed graph read falls back to recompute and never blocks a retry. Schema migration is part of the work. Tests (common work): first attempt writes the graph; retry reuses it without recompute; payload version change invalidates and recomputes; graph read failure falls back to recompute. No timing metrics (option C not chosen). No other choice changed.\nHistory: none\n\n### T1: TODO \u2014 dead-letter retention / purge policy\nFinding: Section 4 finding 2, P3, confidence 5/10, no plan line (gap in D3's approved dead-letter store), reviewer: plan-eng-review (Claude)\nPlan baseline: D3 approved a dead-letter store with no retention or purge. Nothing approved for T1.\nRuntime evidence: unknown \u2014 store does not exist yet. Growth rate = failed jobs only; slow, unbounded.\nTODO record:\n- **What:** a scheduled purge of dead-letter entries older than a configurable retention (default 90 days), skipping entries flagged keep.\n- **Why:** the store grows without bound; a year of failed jobs becomes a slow query behind the alert and the replay UI.\n- **Pros:** bounded storage; predictable query cost; a clear answer to \"how long do we keep failed jobs.\"\n- **Cons:** purging deletes the only record of lost work; wrong default destroys evidence; one more scheduled job to run.\n- **Context:** dead-letter store is new in this PR (D3); growth only matters after months. Start in the retry-policy module's dead-letter code; add a scheduled job and a config value.\n- **Depends on / blocked by:** D3 dead-letter store shipped; retention period agreed with whoever owns incident evidence.\nComparison grid:\n\n| Choice | Current | A) Add to TODOS.md | B) Skip | C) Build now in this PR |\n|---|---|---|---|---|\n| T1 dead-letter retention | none | tracked TODO with trigger: build when the store passes 10k rows or at 3 months, whichever first | not tracked | scheduled purge job, retention config default 90 days, keep flag, tests, in this PR |\n| R1\u2013R6 | approved (D1\u2013D9) | fixed | fixed | fixed |\n\nQuestion D10:\nD10 \u2014 Track dead-letter retention as a TODO, skip it, or build it now?\nProject/branch/task: main branch, retry-framework plan; follow-up to the dead-letter store approved in D3.\nELI10: The dead-letter store keeps every job that ran out of retries or hit a fatal error. Nothing ever removes them. That is fine for months, then the table is large, the growth alert query slows, and nobody remembers why. A purge job with a retention period fixes it, but the retention period is a judgment call about how long failed-job evidence must stay around.\nStakes if we pick wrong: build it now with the wrong retention and you delete evidence of lost work; skip it and the store becomes an unbounded table someone discovers during an incident.\nRecommendation: A because the store is new, growth is slow, and the retention period deserves an owner's answer rather than a default picked inside a retry PR; the TODO carries a concrete trigger.\nCompleteness: A=6/10, B=2/10, C=10/10\nPros / cons:\nA) Add to TODOS.md with a trigger (recommended)\n \u2705 Keeps this PR right-sized: the retry framework ships without a retention debate attached\n \u2705 Trigger (10k rows or 3 months) means the TODO fires before growth matters (human: ~5 min / CC: ~1 min now)\n \u274c Unbounded growth until someone acts on the TODO; the ceiling is a slow query, not data loss\nB) Skip\n \u2705 Nothing to track or build\n \u2705 Zero effort now\n \u274c The store grows forever with no record that anyone considered it\nC) Build now in this PR\n \u2705 Store ships bounded from day one: purge job, 90-day default, keep flag, tests (human: ~2 hr / CC: ~10 min)\n \u2705 No follow-up to forget\n \u274c Expands this PR with a scheduled job and a retention default nobody has agreed to; deletes evidence if the default is wrong\nNet: A tracked follow-up with a trigger, versus a bigger PR that guesses how long failed-job evidence should live.\nHeader: DLQ retention\nOptions:\nA) Add to TODOS.md (recommended)\nRecord the TODO (what/why/pros/cons/context/depends-on) with trigger: build when the dead-letter store passes 10k rows or at 3 months, whichever first. Human ~5 min / CC ~1 min. Completeness 6/10.\nB) Skip \u2014 not valuable enough\nDo not track retention. Completeness 2/10.\nC) Build it now in this PR\nScheduled purge job, retention config default 90 days, keep flag, tests, shipped with the dead-letter store. Human ~2 hr / CC ~10 min. Completeness 10/10.\n\nState: approved\nActual answer: A) Add to TODOS.md \u2014 user answer to D10 (accepted shortcut, completeness 6/10; logged with ceiling and trigger)\nAccepted scope: TODO \"dead-letter retention / purge policy\" with the record above, trigger: build when the dead-letter store passes 10k rows or at 3 months after the store ships (2026-12-29 at the latest if it ships now), whichever first. Ceiling: unbounded table growth until then; slow query, no data loss. TODOS.md is not writable in plan mode: content presented as **not persisted** in the report. When implementing D3, mark the dead-letter persist site with `gstack-shortcut(dec-): unbounded growth, upgrade when store > 10k rows or 3 months`. No implementation approved.\nHistory: none\n\n### T2: TODO \u2014 stable webhook event id header and opt-in at-least-once delivery\nFinding: Section 1 finding 1 follow-up (D2 chose at-most-once), P3, confidence 6/10, PLAN.md:17-19, reviewer: plan-eng-review (Claude)\nPlan baseline: D2 approved at-most-once for `processWebhookJob()`; post-send failures drop the event into dead-letter. Nothing approved for T2.\nRuntime evidence: unknown \u2014 webhook payload/headers not in checkout. Industry practice: stable per-event id header (Stripe `id`, Shopify `X-Shopify-Webhook-Id`, Svix `webhook-id`) plus at-least-once delivery.\nTODO record:\n- **What:** add a stable per-event id header to every webhook delivery, then offer receivers an opt-in at-least-once mode (retry post-send failures) once they dedupe on that id.\n- **Why:** under D2, every timeout or 5xx after send still loses the event; the standard cure is at-least-once with an idempotency key, which D2 declined for now.\n- **Pros:** closes the remaining webhook loss path; matches what receivers expect from major providers; the header alone is harmless under at-most-once.\n- **Cons:** contract change requiring receiver communication and docs; per-receiver opt-in adds a config dimension to the webhook worker.\n- **Context:** start from the D2 decision record and the dead-letter entries with post-send failure class; those counts show how much is being lost. Header first, semantics second.\n- **Depends on / blocked by:** D2 decision (would be superseded for opted-in receivers); D3 dead-letter store for measuring loss; receiver docs channel.\nComparison grid:\n\n| Choice | Current | A) Add to TODOS.md | B) Skip | C) Ship the header now, semantics later |\n|---|---|---|---|---|\n| T2 webhook event id + at-least-once opt-in | none | tracked TODO with trigger: post-send dead-letter entries exceed 1% of webhook sends in any week, or a receiver requests redelivery | not tracked | stable event id header added to every webhook in this PR (no semantics change, D2 intact); at-least-once opt-in stays a TODO |\n| R2 webhook delivery | approved A (D2) | fixed | fixed | fixed (header only, no retry-after-send) |\n| R1, R3\u2013R6, T1 | approved (D1, D3\u2013D10) | fixed | fixed | fixed |\n\nQuestion D11:\nD11 \u2014 Track the webhook event-id header and at-least-once opt-in as a TODO, skip it, or ship the header now?\nProject/branch/task: main branch, retry-framework plan; follow-up to D2 (webhook worker stays at-most-once).\nELI10: D2 kept the promise that a webhook is never sent twice, which means a send that times out is still lost. The usual fix is to stamp every event with a stable id so receivers can ignore duplicates, and then retry freely. That is a contract change, so it was declined for this PR. The question is whether to track it, drop it, or at least ship the harmless id header now so receivers can start deduping before the semantics ever change.\nStakes if we pick wrong: lost webhook events keep landing in dead-letter with no plan to stop the loss; or a header change rides along in a retry PR without receiver communication.\nRecommendation: A because this is a receiver-facing contract change that deserves its own PR and docs, and the dead-letter store (D3) will produce the loss numbers that justify it; the trigger is concrete.\nCompleteness: A=6/10, B=2/10, C=8/10\nPros / cons:\nA) Add to TODOS.md with a trigger (recommended)\n \u2705 Keeps the retry PR free of webhook contract changes; the TODO fires on measured loss (human: ~5 min / CC: ~1 min now)\n \u2705 Dead-letter counts of post-send failures give the case for it with real numbers\n \u274c Webhook events lost to timeouts stay lost until the TODO is acted on\nB) Skip\n \u2705 Nothing to track\n \u2705 Zero effort now\n \u274c No record that at-most-once was a deliberate trade with a known cost\nC) Ship the stable event-id header now, at-least-once later\n \u2705 Receivers can start deduping today; the header is harmless under at-most-once (human: ~2 hr / CC: ~10 min)\n \u2705 Makes the eventual semantics change a config flip instead of a payload change\n \u274c Adds a webhook payload change and receiver docs to a retry PR; still needs the TODO for the semantics\nNet: Track it with a loss-based trigger, or ship a small header change now inside a PR about retries.\nHeader: Webhook TODO\nOptions:\nA) Add to TODOS.md (recommended)\nRecord the TODO with trigger: post-send dead-letter entries exceed 1% of webhook sends in any week, or a receiver requests redelivery. Human ~5 min / CC ~1 min. Completeness 6/10.\nB) Skip \u2014 not valuable enough\nDo not track. Completeness 2/10.\nC) Ship the header now\nAdd a stable per-event id header to every webhook delivery in this PR; D2 semantics unchanged; at-least-once opt-in remains a TODO. Human ~2 hr / CC ~10 min. Completeness 8/10.\n\nState: approved\nActual answer: A) Add to TODOS.md \u2014 user answer to D11 (accepted shortcut, completeness 6/10; logged with ceiling and trigger)\nAccepted scope: TODO \"stable webhook event id header + opt-in at-least-once delivery\" with the record above, trigger: post-send dead-letter entries exceed 1% of webhook sends in any week, or a receiver requests redelivery. Ceiling: webhook events lost to post-send failures stay lost (parked in dead-letter, D3). TODOS.md not writable in plan mode: content presented as **not persisted**. No implementation approved; D2 unchanged.\nHistory: none\n\n**Approval readiness: PASS** \u2014 checked R1 (D1=A), R2 (D2=A), R3a (D3=A), R3b (D4=A), R3c (D5=A), R3d (D6=A), R4 (D7=A), R5 (D8=A), R6 (D9=A), T1 (D10=A), T2 (D11=A). Every accepted remedy cites its own user answer; the R5 regression contract is approved with explicit assertions; no deferral is unresolved. Total elapsed retry window: considered, not a choice (bounded by D3 maxAttempts \u00d7 D5 maxDelay \u2248 50 min at defaults).\n\n## Working plan (after review)\n\n1. **Mechanism (D1):** all 5 workers retry through the job library's built-in retry hooks. No custom inline scheduler. Reopen only if the hook API cannot take a custom curve callback.\n2. **Shared retry-policy module (D7):** `buildBackoff`, `classify`, `toDeadLetter`, `logAttempt`, `validateConfig`; retry state-machine diagram inline. Each worker supplies its values only.\n3. **Curve (D4, D5):** full jitter, delay = min(random(0, base\u00b72^n), maxDelay); defaults base 1 s, multiplier 2, maxDelay 10 min; injectable RNG; per-worker override.\n4. **Terminal handling (D3):** maxAttempts default 5, per-worker override; exhaustion or fatal error \u2192 dead-letter store (last error, attempt history, payload ref); metric + alert on growth; manual replay; persist failure keeps the job failed and logs both errors.\n5. **Error classes (D6):** 4 non-webhook workers declare retryable and fatal classes; fatal \u2192 dead-letter immediately; unknown \u2192 retryable.\n6. **Webhook worker (D2):** at-most-once preserved. Retry only pre-send failures (refused, DNS, TLS, local). Post-send timeout/5xx/reset \u2192 terminal, dead-letter, no re-send. Webhook replay from dead-letter warns it may duplicate.\n7. **Regression proof (D8):** unit tests with a recording fake transport per failure class + integration through the real hooks with a fake receiver and a simulated restart. CRITICAL.\n8. **Graph persistence (D9):** compute once on attempt 1, persist beside the job with payload version; reuse on retry; invalidate on version change; read failure \u2192 recompute. Schema migration + 4 tests.\n9. **TODOs (D10, D11):** dead-letter retention; webhook event-id header + at-least-once opt-in. Not persisted (plan mode).\n\n## NOT in scope\n- **Idempotency key / at-least-once webhooks:** deferred to TODO (D11); D2 chose at-most-once for this PR.\n- **Dead-letter retention / purge:** deferred to TODO (D10); store is new and growth is slow.\n- **Total elapsed retry window:** not needed; D3 \u00d7 D5 bounds the window at ~50 min with defaults.\n- **Graph timing metrics (R6 option C):** not chosen; D9 persists the graph instead.\n- **Custom inline scheduler (PLAN.md:7-9):** replaced by library hooks (D1).\n- **Per-receiver circuit breaker for webhooks:** not raised in the plan; separate scope if post-send failures cluster by receiver.\n\n## What already exists\n- **Reused:** the job library's retry hooks and attempt-count persistence (D1). Unverified in this checkout; the plan itself states the library version has the same shape (PLAN.md:8).\n- **Assumed existing, unverified:** structured logging and a metrics/alerting pipeline for the dead-letter growth alert (D3).\n- **New:** retry-policy module (D7, shared-code rubric evidence in R4), dead-letter store + replay path (D3), graph persistence column/table + migration (D9).\n- **Rebuilt:** nothing. The custom scheduler from the original plan is dropped.\n\n## Diagrams\n- Retry state machine: Section 1 above; goes inline in the retry-policy module (D7).\n- Coverage diagram: Section 3 above.\n- Files needing inline diagrams once written: the retry-policy module (state machine); the webhook worker (pre-send vs post-send failure split, D2).\n\n## Failure modes\n| New path | Realistic production failure | Test coverage (approved) | Error handling (approved) | User-visible? |\n|---|---|---|---|---|\n| Hook wiring (D1) in each worker | hook registered wrong \u2192 no retries ever happen | integration test per worker (Section 3), webhook integration (D8) | n/a, caught by tests | silent without tests \u2192 covered |\n| Dead-letter persist (D3) | dead-letter store down while a job exhausts | unit test: persist failure path | job stays failed; both errors logged; metric | operator sees log + metric |\n| Webhook receiver down 2 h (D2) | every send times out after write | unit + integration per failure class | terminal \u2192 dead-letter, no re-send; alert on growth | operator alert; receiver gets no duplicate |\n| Poisoned job (D3, D6) | throws the same error forever | exhaustion test; fatal-class test | fatal \u2192 dead-letter on attempt 1; else at maxAttempts | dead-letter entry with cause |\n| Graph read (D9) | stored graph corrupt or missing | fallback test | recompute, never block the retry | none |\n| Worker crash mid-curve (D1) | process killed between attempts | integration restart test (D8) | library persists attempt count | none |\n| Config error (D7) | maxAttempts 0 or maxDelay < base | validateConfig tests | fail fast at startup | deploy fails loudly |\n\n**Critical gaps flagged: 0** (every new path has both approved test coverage and approved error handling).\n\n## Worktree parallelization strategy\n\n| Step | Modules touched | Depends on |\n|------|----------------|------------|\n| S1 retry-policy module + unit tests | retry-policy (new) | \u2014 |\n| S2 dead-letter store, metric, alert, replay | dead-letter store (new), migrations, metrics | \u2014 (agree `toDeadLetter` signature with S1 first) |\n| S3 graph persistence + migration + tests | job storage / migrations, graph builder | \u2014 |\n| S4 wire 4 non-webhook workers + integration tests | workers/, retry-policy | S1, S2 |\n| S5 webhook worker rewrite + regression tests | workers/ (webhook), retry-policy | S1, S2 |\n| S6 config defaults, docs, inline diagram | retry-policy, docs | S1 |\n\n**Parallel lanes:** Lane A: S1 \u2192 S4 + S5 + S6 (S4/S5 touch disjoint worker files) \u00b7 Lane B: S2 \u00b7 Lane C: S3.\n**Execution order:** Launch A(S1) + B + C. Merge all three. Then S4, S5, S6 in parallel. Merge.\n**Conflict flags:** S1/S2 share the `toDeadLetter` contract \u2014 fix the signature before launching. S3 and S4/S5 both touch each worker's job-load path \u2014 land S3 first or coordinate the graph-load call site.\n\n## Implementation Tasks\nSynthesized from this review's findings. Each task derives from a specific finding above. Run with Claude Code or Codex; checkbox as you ship.\n\n- [ ] **T1 (P1, human: ~1 day / CC: ~20 min)** \u2014 retry-policy module \u2014 Build `buildBackoff` (full jitter, cap, defaults), `classify`, `toDeadLetter` (persist + metric + persist-failure handling), `logAttempt`, `validateConfig`, inline state-machine diagram, unit tests for every branch\n - Surfaced by: Code quality \u2014 R4/D7 \"duplicated across 5 worker files\"; Section 1 \u2014 R3b/R3c/R3d\n - Files: retry-policy module (new; path not in checkout)\n - Verify: module unit suite green: jitter bounds, cap at high n, defaults, class directions, unknown \u2192 retryable, persist-failure path, config validation\n- [ ] **T2 (P1, human: ~1.5 days / CC: ~30 min)** \u2014 dead-letter store \u2014 Schema + write path (last error, attempt history, payload ref), growth counter metric + alert, manual replay path, webhook replay \"may duplicate\" warning, `gstack-shortcut` marker for retention (D10)\n - Surfaced by: Architecture \u2014 R3a/D3 \"no attempt limit or terminal outcome\"; Performance \u2014 alert reads a counter, not COUNT(*)\n - Files: dead-letter store + migration (new), metrics config\n - Verify: exhaustion \u2192 entry; fatal \u2192 entry on attempt 1; alert fires on growth; replay re-enqueues once; webhook replay shows warning\n- [ ] **T3 (P1, human: ~1.5 days / CC: ~30 min)** \u2014 webhook worker \u2014 Rewrite `processWebhookJob()` on library hooks with pre-send vs post-send classification; at-most-once preserved; CRITICAL regression tests (unit fake transport + integration fake receiver + restart)\n - Surfaced by: Tests \u2014 R5/D8 \"No regression test for the prior at-most-once delivery guarantee\"; Architecture \u2014 R2/D2\n - Files: workers/ webhook worker (path not in checkout), its tests\n - Verify: exactly 1 send after pre-send retries; 0 further sends after timeout/5xx/reset with dead-letter entry; attempt count survives restart\n- [ ] **T4 (P1, human: ~1.5 days / CC: ~30 min)** \u2014 4 non-webhook workers \u2014 Wire each to library hooks with module config (attempts, maxDelay, retryable/fatal lists); remove inline envelopes; one integration test per worker\n - Surfaced by: Scope Challenge \u2014 R1/D1 \"roll a custom exponential-backoff scheduler inline\"; Section 1 \u2014 R3d/D6\n - Files: workers/ (4 files; paths not in checkout)\n - Verify: transient \u2192 retried with module delay; fatal \u2192 dead-letter, no attempts consumed; exhaustion \u2192 dead-letter; deleted job row \u2192 fatal\n- [ ] **T5 (P2, human: ~1 day / CC: ~20 min)** \u2014 graph persistence \u2014 Migration for graph + payload version beside the job; store on attempt 1; reuse on retry; invalidate on version change; read failure \u2192 recompute\n - Surfaced by: Performance \u2014 R6/D9 \"re-fetch the full job payload ... recompute the dependency graph\"\n - Files: job storage migration (new), graph builder call site in workers/\n - Verify: 4 tests: first-attempt write, retry reuse, version invalidation, read-failure fallback\n- [ ] **T6 (P2, human: ~2 hr / CC: ~10 min)** \u2014 config + docs \u2014 Document defaults (base 1 s, \u00d72, maxDelay 10 min, maxAttempts 5) and per-worker overrides; document at-most-once webhook semantics and replay caveat\n - Surfaced by: Architecture \u2014 R3a\u2013R3c/D3\u2013D5; R2/D2\n - Files: retry-policy module docs, worker config, README/runbook\n - Verify: docs review; `validateConfig` rejects bad values at startup\n- [ ] **T7 (P3, human: ~10 min / CC: ~2 min)** \u2014 TODOS.md \u2014 Add the two accepted TODOs (dead-letter retention; webhook event-id header + at-least-once opt-in) with triggers\n - Surfaced by: TODOS.md updates \u2014 D10, D11 (not persisted in plan mode)\n - Files: TODOS.md (new)\n - Verify: entries present with What/Why/Pros/Cons/Context/Depends-on\n\nEffort ratios assumed: features ~30x, tests ~50x, bug fix with regression ~20x, architecture ~5x.\n\n## TODOS.md content (accepted, **not persisted** \u2014 plan mode forbids repo writes)\n1. **Dead-letter retention / purge policy** \u2014 record in ledger T1 (D10). Trigger: store > 10k rows or 3 months after ship.\n2. **Stable webhook event-id header + opt-in at-least-once** \u2014 record in ledger T2 (D11). Trigger: post-send dead-letter entries > 1% of webhook sends in any week, or a receiver requests redelivery.\n\n## Unresolved decisions that may bite you later\nNone. All 11 decisions (D1\u2013D11) answered.\n\n## Suppressed findings (appendix, confidence \u2264 4)\n- `[P3] (confidence: 4/10)` \u2014 total elapsed retry window bound; bounded in practice by D3 \u00d7 D5. Not promoted.\n- `[P3] (confidence: 3/10)` \u2014 per-receiver circuit breaker for the webhook worker; no evidence of receiver clustering in the plan. Listed under NOT in scope.\n\n## Completion summary\n- Step 0: Scope Challenge \u2014 scope accepted as-is (mechanism changed to library hooks per D1; no feature cut)\n- Architecture Review: 3 issues found\n- Code Quality Review: 4 issues found\n- Test Review: diagram produced, 27 gaps identified\n- Performance Review: 3 issues found\n- NOT in scope: written\n- What already exists: written\n- TODOS.md updates: 2 items proposed to user (both accepted; not persisted)\n- Failure modes: 0 critical gaps flagged\n- Unresolved decisions: 0 in this review\n- Outside voice: provider codex, disabled (codex_reviews disabled; no native replacement dispatched)\n- Parallelization: 3 lanes, 3 parallel / 2 sequential steps (S1 \u2192 S4/S5/S6)\n- Lake Score: 4/8 = 10/10 choices / answered coverage choices (D3, D6, D7, D8 at 10/10; D1, D5 at 9/10; D10, D11 at 6/10 accepted shortcuts; D2, D4, D9 were kind choices, excluded)\n- Test Plan Artifact: `~/.gstack/projects/gstack-plan-count-vrYrwf/user-main-eng-review-test-plan-20260929-200336.md`\n- Implementation Tasks JSONL: `~/.gstack/projects/gstack-plan-count-vrYrwf/tasks-eng-review-20260929-200911.jsonl` (7 tasks)\n\n## GSTACK REVIEW REPORT\n\n| Review | Trigger | Why | Runs | Status | Findings |\n|--------|---------|-----|------|--------|----------|\n| CEO Review | `/plan-ceo-review` | Scope & strategy | 0 | \u2014 | \u2014 |\n| Outside Review | codex via `/plan-eng-review` Outside Voice (host: claude, phase: plan-review) | Independent 2nd opinion | 1 | DISABLED | skipped \u2014 codex_reviews disabled |\n| Eng Review | `/plan-eng-review` | Architecture & tests (required) | 1 | ISSUES OPEN | 37 issues, 0 critical gaps (this run; logged at finish step 3) |\n| Design Review | `/plan-design-review` | UI/UX gaps | 0 | \u2014 | \u2014 |\n| DX Review | `/plan-devex-review` | Developer experience gaps | 0 | \u2014 | \u2014 |\n\n**OUTSIDE COVERAGE:** provider codex, phase plan-review, outside_status disabled (codex_reviews disabled in gstack config), no findings; no native replacement dispatched. Re-enable: `gstack-config set codex_reviews enabled`.\n\n**VERDICT:** No reviews CLEARED. Eng Review is ISSUES OPEN: 37 mapped issues (3 architecture, 4 code quality, 27 test gaps, 3 performance), all resolved into decisions D1\u2013D11 and tasks T1\u2013T7, 0 critical gaps \u2014 eng review required.\n\nNO UNRESOLVED DECISIONS\n" +} \ No newline at end of file diff --git a/test/fixtures/eng-batching-unsourced-brief-36606688266.json b/test/fixtures/eng-batching-unsourced-brief-36606688266.json new file mode 100644 index 000000000..661f0fc2e --- /dev/null +++ b/test/fixtures/eng-batching-unsourced-brief-36606688266.json @@ -0,0 +1,311 @@ +{ + "source": "run 36606688266 plan-eng-multi-finding-batching observation.json (transcript.calls) and the saved report replayed from its Write/Edit inputs", + "recordedOutcome": { + "outcome": "completion_summary", + "step0Count": 10, + "reviewCount": 0 + }, + "calls": [ + { + "sessionId": "cd04b55a-5f1a-4672-b3ed-40ea1b1636bc", + "toolUseId": "toolu_011vfjpt7sgeSRMw1zhHATN8", + "questions": [ + { + "question": "D1 — Use the job library's retry hooks or roll a custom inline scheduler?\nProject/branch/task: `main` of the plan fixture repo, plan \"Add background job retry framework\".\nELI10: The job library you already run has a retry system built in. The plan wants to rebuild the same thing by hand inside each of the 5 workers, only so the delay curve can be tuned. Retry systems are easy to get subtly wrong: losing the attempt count when a worker restarts, retrying forever, or retrying twice at once. The library has already paid for those bugs; a hand-rolled copy pays for them again in production.\nStakes if we pick wrong: a hand-rolled scheduler that drops attempt state on restart or double-schedules turns one failed job into duplicate side effects or an infinite retry loop, with no dead-letter to catch it.\nRecommendation: A because the plan admits the shapes are identical, and the curve is pluggable in the library through a strategy function, so A delivers the same control with far less new code. (human: ~1 day / CC: ~20 min for A; human: ~1 week / CC: ~2 h for B plus ongoing ownership)\nCompleteness: A=9/10, B=5/10\nPros / cons:\nA) Library hooks + custom curve (recommended)\n ✅ Attempt counting, persistence across restarts and terminal handling come from tested library code, not new code\n ✅ The custom curve still lives in one strategy function, so \"full control over the curve\" is preserved\n ❌ Requires confirming the library's hook accepts a custom delay function; if it does not, we fall back to B for the curve only\nB) Custom inline scheduler\n ✅ Zero dependency on the library's retry semantics or its upgrade cadence\n ✅ Any curve shape, any bookkeeping, no hook constraints\n ❌ Rebuilds attempt state, restart persistence, concurrency guards and dead-lettering by hand, and those are the parts that fail at 3am\nNet: you are trading a one-line strategy function against owning a second retry engine forever.", + "header": "Retry engine", + "multiSelect": false, + "options": [ + { + "label": "Library hooks + custom curve (recommended)", + "description": "Register the exponential-backoff curve as one custom backoff strategy function with the job library's built-in retry hooks. Attempt counting, persistence across worker restarts, and terminal/dead-letter handling come from the library. Verify the hook accepts a delay function first; if it does not, fall back to a custom curve only for delay computation while keeping library scheduling. R2-R5 stay pending. Completeness 9/10. human: ~1 day / CC: ~20 min." + }, + { + "label": "Custom inline scheduler", + "description": "Keep the plan as written: a custom exponential-backoff scheduler inline in each worker, bypassing the library's retry hooks. Attempt state, restart persistence, concurrency guards and terminal handling must be designed and tested by hand (tracked under R3). R2-R5 stay pending. Completeness 5/10. human: ~1 week / CC: ~2 h plus ongoing ownership." + } + ] + } + ], + "answered": true, + "failed": false, + "answers": { + "D1 — Use the job library's retry hooks or roll a custom inline scheduler?\nProject/branch/task: `main` of the plan fixture repo, plan \"Add background job retry framework\".\nELI10: The job library you already run has a retry system built in. The plan wants to rebuild the same thing by hand inside each of the 5 workers, only so the delay curve can be tuned. Retry systems are easy to get subtly wrong: losing the attempt count when a worker restarts, retrying forever, or retrying twice at once. The library has already paid for those bugs; a hand-rolled copy pays for them again in production.\nStakes if we pick wrong: a hand-rolled scheduler that drops attempt state on restart or double-schedules turns one failed job into duplicate side effects or an infinite retry loop, with no dead-letter to catch it.\nRecommendation: A because the plan admits the shapes are identical, and the curve is pluggable in the library through a strategy function, so A delivers the same control with far less new code. (human: ~1 day / CC: ~20 min for A; human: ~1 week / CC: ~2 h for B plus ongoing ownership)\nCompleteness: A=9/10, B=5/10\nPros / cons:\nA) Library hooks + custom curve (recommended)\n ✅ Attempt counting, persistence across restarts and terminal handling come from tested library code, not new code\n ✅ The custom curve still lives in one strategy function, so \"full control over the curve\" is preserved\n ❌ Requires confirming the library's hook accepts a custom delay function; if it does not, we fall back to B for the curve only\nB) Custom inline scheduler\n ✅ Zero dependency on the library's retry semantics or its upgrade cadence\n ✅ Any curve shape, any bookkeeping, no hook constraints\n ❌ Rebuilds attempt state, restart persistence, concurrency guards and dead-lettering by hand, and those are the parts that fail at 3am\nNet: you are trading a one-line strategy function against owning a second retry engine forever.": "Library hooks + custom curve (recommended)" + }, + "unansweredQuestionIndices": [], + "answeredAt": "2026-09-29T17:46:48.252Z" + }, + { + "sessionId": "cd04b55a-5f1a-4672-b3ed-40ea1b1636bc", + "toolUseId": "toolu_0182pLbYfj4VpsRE57iBzMYQ", + "questions": [ + { + "question": "D2 — What delivery guarantee does `processWebhookJob()` keep once it retries?\nProject/branch/task: `main` of the plan fixture repo, plan \"Add background job retry framework\".\nELI10: Today a webhook is sent once; if it fails, it is gone (at-most-once). The moment you retry, a request that timed out after the customer already received it gets sent again, so the customer sees the same event twice. You have to pick: either only retry when you are sure the request never left, or retry freely but stamp every attempt with the same id so the customer can ignore repeats. The plan does neither and just retries.\nStakes if we pick wrong: customers process duplicate events (double orders, double emails) with no way to detect them, or you ship a retry feature that almost never fires because most webhook failures are timeouts.\nRecommendation: B because it is the standard webhook contract (retry on timeout/5xx, stable event id per attempt) and is the only option where retrying actually improves delivery while giving receivers a way to dedupe. This is a receiver-visible contract change; A is the right pick if you cannot communicate it to receivers.\nCompleteness: A=7/10, B=9/10, C=3/10\nPros / cons:\nA) Keep at-most-once\n ✅ No change to what receivers see; the existing guarantee and its regression test stay valid as-is\n ✅ Smallest blast radius: no new headers, no receiver communication needed\n ❌ Retries only fire on connect/DNS/pre-send errors; timeouts and 5xx go straight to terminal, so most real failures are still not retried\nB) At-least-once + idempotency key (recommended)\n ✅ Timeouts and 5xx are retried, so delivery reliability actually improves for receivers\n ✅ Same delivery id on every attempt lets receivers dedupe; this is the contract Stripe/GitHub-style webhooks use\n ❌ Receiver-visible contract change: duplicates become possible and receivers must be told to dedupe on the id\nC) Plain retry (plan as written)\n ✅ Least code: no header, no classification, just retry on any failure\n ✅ Ships fastest\n ❌ Duplicates reach receivers with no way to tell them apart; silent double side effects\nNet: A keeps the promise but retries little; B changes the promise but makes retries worth having; C breaks the promise silently.", + "header": "Webhook semantics", + "multiSelect": false, + "options": [ + { + "label": "Keep at-most-once", + "description": "Preserve at-most-once for `processWebhookJob()`. Retry only failures where the request provably never left the process (connection refused, DNS failure, errors before send). Timeouts, 5xx and any post-send ambiguity are terminal and go to whatever terminal handling R3 decides. No new headers; receiver contract unchanged. R3, R4, R5, R7 stay pending. Completeness 7/10. human: ~half day / CC: ~10 min." + }, + { + "label": "At-least-once + idempotency key (recommended)", + "description": "Retry timeouts and 5xx as well as pre-send failures. Every attempt carries the same stable delivery id (e.g. an `Idempotency-Key`/event-id header derived from the job id, generated once, persisted with the job). 4xx other than 408/429 is non-retryable for webhooks. Document to receivers that duplicates are possible and must be deduped on the id. R3, R4, R5, R7 stay pending. Completeness 9/10. human: ~1 day / CC: ~20 min." + }, + { + "label": "Plain retry (plan as written)", + "description": "Retry `processWebhookJob()` on any failure with no delivery id and no classification. Duplicates reach receivers undetectably. R3, R4, R5, R7 stay pending. Completeness 3/10. human: ~1 h / CC: ~5 min." + } + ] + } + ], + "answered": true, + "failed": false, + "answers": { + "D2 — What delivery guarantee does `processWebhookJob()` keep once it retries?\nProject/branch/task: `main` of the plan fixture repo, plan \"Add background job retry framework\".\nELI10: Today a webhook is sent once; if it fails, it is gone (at-most-once). The moment you retry, a request that timed out after the customer already received it gets sent again, so the customer sees the same event twice. You have to pick: either only retry when you are sure the request never left, or retry freely but stamp every attempt with the same id so the customer can ignore repeats. The plan does neither and just retries.\nStakes if we pick wrong: customers process duplicate events (double orders, double emails) with no way to detect them, or you ship a retry feature that almost never fires because most webhook failures are timeouts.\nRecommendation: B because it is the standard webhook contract (retry on timeout/5xx, stable event id per attempt) and is the only option where retrying actually improves delivery while giving receivers a way to dedupe. This is a receiver-visible contract change; A is the right pick if you cannot communicate it to receivers.\nCompleteness: A=7/10, B=9/10, C=3/10\nPros / cons:\nA) Keep at-most-once\n ✅ No change to what receivers see; the existing guarantee and its regression test stay valid as-is\n ✅ Smallest blast radius: no new headers, no receiver communication needed\n ❌ Retries only fire on connect/DNS/pre-send errors; timeouts and 5xx go straight to terminal, so most real failures are still not retried\nB) At-least-once + idempotency key (recommended)\n ✅ Timeouts and 5xx are retried, so delivery reliability actually improves for receivers\n ✅ Same delivery id on every attempt lets receivers dedupe; this is the contract Stripe/GitHub-style webhooks use\n ❌ Receiver-visible contract change: duplicates become possible and receivers must be told to dedupe on the id\nC) Plain retry (plan as written)\n ✅ Least code: no header, no classification, just retry on any failure\n ✅ Ships fastest\n ❌ Duplicates reach receivers with no way to tell them apart; silent double side effects\nNet: A keeps the promise but retries little; B changes the promise but makes retries worth having; C breaks the promise silently.": "Keep at-most-once" + }, + "unansweredQuestionIndices": [], + "answeredAt": "2026-09-29T17:48:33.604Z" + }, + { + "sessionId": "cd04b55a-5f1a-4672-b3ed-40ea1b1636bc", + "toolUseId": "toolu_012ueyTcioa6A2y9YE4gYPcR", + "questions": [ + { + "question": "D3 — How many times may a job retry, and where does it go when it gives up?\nProject/branch/task: `main` of the plan fixture repo, plan \"Add background job retry framework\".\nELI10: A retry curve without a stopping point is a job that runs forever when the thing it depends on is down for good. You need a maximum number of tries, and you need a place for jobs that used up their tries (a dead-letter set) so someone can look at them and replay them later. Otherwise failed work quietly disappears or quietly never stops.\nStakes if we pick wrong: either a poisoned job hammers a downstream forever and starves healthy jobs, or failed webhooks and jobs vanish with only a log line nobody reads.\nRecommendation: A because the library already provides the failed set, so the dead-letter and alert cost a config line and one log call, and it is the only option where an operator can find and replay a lost job.\nCompleteness: A=9/10, B=6/10, C=2/10\nPros / cons:\nA) Bounded + dead-letter + alert (recommended)\n ✅ Exhausted jobs are inspectable and replayable from the library's failed set, with the last error attached\n ✅ One structured log line plus a metric on dead-letter entry makes a downstream outage visible within minutes\n ❌ Needs a per-worker ceiling value and a dead-letter retention/cleanup policy to be chosen and documented\nB) Bounded + log-and-drop\n ✅ Bounds the retry loop with the least configuration\n ✅ No dead-letter retention to manage\n ❌ A dropped job is gone; the only trace is a log line, so replay after an outage is impossible\nC) Unbounded (plan as written)\n ✅ No ceiling to tune; a job eventually succeeds if the dependency ever recovers\n ✅ Zero extra code\n ❌ Permanently failing jobs retry forever, consume worker capacity and never surface as a problem\nNet: you are choosing whether a job that cannot succeed becomes a visible artifact, a log line, or a permanent background load.", + "header": "Attempt ceiling", + "multiSelect": false, + "options": [ + { + "label": "Bounded + dead-letter + alert (recommended)", + "description": "Set a maximum attempt count per worker (default 5, overridable per worker, configured in the same place as the backoff strategy). On exhaustion or on a non-retryable error, the job lands in the library's dead-letter/failed set with its last error; emit one structured error log and a metric on entry. Webhook timeouts/5xx (terminal per R2) land here too. Document the retention/replay procedure. R4, R5 stay pending. Completeness 9/10. human: ~half day / CC: ~15 min." + }, + { + "label": "Bounded + log-and-drop", + "description": "Set the same per-worker maximum attempt count (default 5). On exhaustion, log the error and drop the job; no dead-letter set, no metric, no replay. R4, R5 stay pending. Completeness 6/10. human: ~1 h / CC: ~5 min." + }, + { + "label": "Unbounded (plan as written)", + "description": "No attempt ceiling; the exponential curve continues until the job succeeds. No terminal handling. Conflicts with R2, which needs a terminal destination for webhook timeouts. R4, R5 stay pending. Completeness 2/10. human: 0 / CC: 0." + } + ] + } + ], + "answered": true, + "failed": false, + "answers": { + "D3 — How many times may a job retry, and where does it go when it gives up?\nProject/branch/task: `main` of the plan fixture repo, plan \"Add background job retry framework\".\nELI10: A retry curve without a stopping point is a job that runs forever when the thing it depends on is down for good. You need a maximum number of tries, and you need a place for jobs that used up their tries (a dead-letter set) so someone can look at them and replay them later. Otherwise failed work quietly disappears or quietly never stops.\nStakes if we pick wrong: either a poisoned job hammers a downstream forever and starves healthy jobs, or failed webhooks and jobs vanish with only a log line nobody reads.\nRecommendation: A because the library already provides the failed set, so the dead-letter and alert cost a config line and one log call, and it is the only option where an operator can find and replay a lost job.\nCompleteness: A=9/10, B=6/10, C=2/10\nPros / cons:\nA) Bounded + dead-letter + alert (recommended)\n ✅ Exhausted jobs are inspectable and replayable from the library's failed set, with the last error attached\n ✅ One structured log line plus a metric on dead-letter entry makes a downstream outage visible within minutes\n ❌ Needs a per-worker ceiling value and a dead-letter retention/cleanup policy to be chosen and documented\nB) Bounded + log-and-drop\n ✅ Bounds the retry loop with the least configuration\n ✅ No dead-letter retention to manage\n ❌ A dropped job is gone; the only trace is a log line, so replay after an outage is impossible\nC) Unbounded (plan as written)\n ✅ No ceiling to tune; a job eventually succeeds if the dependency ever recovers\n ✅ Zero extra code\n ❌ Permanently failing jobs retry forever, consume worker capacity and never surface as a problem\nNet: you are choosing whether a job that cannot succeed becomes a visible artifact, a log line, or a permanent background load.": "Bounded + dead-letter + alert (recommended)" + }, + "unansweredQuestionIndices": [], + "answeredAt": "2026-09-29T17:50:10.303Z" + }, + { + "sessionId": "cd04b55a-5f1a-4672-b3ed-40ea1b1636bc", + "toolUseId": "toolu_016fgg5mWHBjFGWxQPRz9Whr", + "questions": [ + { + "question": "D4 — Add jitter to the backoff curve, or keep it deterministic?\nProject/branch/task: `main` of the plan fixture repo, plan \"Add background job retry framework\".\nELI10: When a downstream service goes down, hundreds of jobs across all 5 workers fail at the same moment. With a pure exponential curve they all wake up at exactly the same moment too, and hit the recovering service as one wave, which can knock it over again. Jitter adds a random spread to each delay so the retries trickle back instead of stampeding.\nStakes if we pick wrong: a downstream that recovers from an outage gets re-flattened by your own synchronized retry wave, turning a 2-minute blip into a 20-minute incident.\nRecommendation: A because it is two lines inside the strategy function you already own and it is the standard mitigation for retry storms; deterministic curves are only useful in tests, which can seed or stub the random source.\nCompleteness: A=9/10, B=6/10\nPros / cons:\nA) Equal jitter (recommended)\n ✅ Retries after a shared outage spread across the window instead of returning as one synchronized burst\n ✅ Lives inside the single strategy function from D1, so every worker gets it with no per-worker code\n ❌ Curve tests need an injectable random source to stay deterministic\nB) No jitter (pure curve)\n ✅ Exact, predictable retry times that are easy to reason about and assert in tests\n ✅ Zero extra code beyond the curve itself\n ❌ All jobs that fail together retry together, so the retry framework itself becomes a traffic amplifier during outages\nNet: predictability in tests against stampede protection in production; the test cost is one injected random source.", + "header": "Jitter", + "multiSelect": false, + "options": [ + { + "label": "Equal jitter (recommended)", + "description": "Inside the single backoff strategy function, compute the exponential delay and return half of it plus a random amount up to the other half (equal jitter). The random source is injectable so tests can pin it. Required proof: unit test that returned delays fall within [curve/2, curve] for each attempt, and that a pinned random source gives a deterministic value. R5 stays pending. Completeness 9/10. human: ~1 h / CC: ~5 min." + }, + { + "label": "No jitter (pure curve)", + "description": "Return the exact exponential curve value with no random spread. Required proof: unit test of the exact value per attempt. R5 stays pending. Completeness 6/10. human: ~30 min / CC: ~3 min." + } + ] + } + ], + "answered": true, + "failed": false, + "answers": { + "D4 — Add jitter to the backoff curve, or keep it deterministic?\nProject/branch/task: `main` of the plan fixture repo, plan \"Add background job retry framework\".\nELI10: When a downstream service goes down, hundreds of jobs across all 5 workers fail at the same moment. With a pure exponential curve they all wake up at exactly the same moment too, and hit the recovering service as one wave, which can knock it over again. Jitter adds a random spread to each delay so the retries trickle back instead of stampeding.\nStakes if we pick wrong: a downstream that recovers from an outage gets re-flattened by your own synchronized retry wave, turning a 2-minute blip into a 20-minute incident.\nRecommendation: A because it is two lines inside the strategy function you already own and it is the standard mitigation for retry storms; deterministic curves are only useful in tests, which can seed or stub the random source.\nCompleteness: A=9/10, B=6/10\nPros / cons:\nA) Equal jitter (recommended)\n ✅ Retries after a shared outage spread across the window instead of returning as one synchronized burst\n ✅ Lives inside the single strategy function from D1, so every worker gets it with no per-worker code\n ❌ Curve tests need an injectable random source to stay deterministic\nB) No jitter (pure curve)\n ✅ Exact, predictable retry times that are easy to reason about and assert in tests\n ✅ Zero extra code beyond the curve itself\n ❌ All jobs that fail together retry together, so the retry framework itself becomes a traffic amplifier during outages\nNet: predictability in tests against stampede protection in production; the test cost is one injected random source.": "Equal jitter (recommended)" + }, + "unansweredQuestionIndices": [], + "answeredAt": "2026-09-29T17:51:07.968Z" + }, + { + "sessionId": "cd04b55a-5f1a-4672-b3ed-40ea1b1636bc", + "toolUseId": "toolu_01AGgMxL2cDbt1vqtMqNz8dH", + "questions": [ + { + "question": "D5 — Should workers name errors that must not be retried, or retry every failure to the ceiling?\nProject/branch/task: `main` of the plan fixture repo, plan \"Add background job retry framework\".\nELI10: Some failures fix themselves if you wait (a database hiccup, a slow API). Others never will (a payload that fails validation, a revoked API key). Retrying the second kind five times with growing delays just wastes capacity and delays the moment someone notices. Letting each worker say \"these error types are permanent\" sends them straight to the dead-letter set on the first try.\nStakes if we pick wrong: a bad payload burns 5 attempts and up to the full backoff window before it surfaces, and during a bad deploy every job does this at once.\nRecommendation: A because it is a small per-worker list, the dead-letter path already exists from D3, and it turns a permanent failure into an immediate signal instead of a delayed one. Medium confidence on which types are permanent; verify against the actual error classes when implementing.\nCompleteness: A=9/10, B=6/10\nPros / cons:\nA) Explicit non-retryable list (recommended)\n ✅ Permanent failures reach the dead-letter set on attempt 1, so operators see bad payloads or revoked credentials within seconds\n ✅ Unknown errors still default to retry, so nothing transient is accidentally dropped\n ❌ Each worker needs a short, reviewed list of permanent error types, and a wrong entry makes a transient error permanent\nB) Retry everything to ceiling\n ✅ No classification to get wrong; behavior is identical for every worker\n ✅ Nothing to maintain when new error types appear\n ❌ Permanent failures consume the full attempt budget and backoff window before anyone can see them\nNet: a short reviewed list per worker against a guaranteed delay on every permanent failure.", + "header": "Error classes", + "multiSelect": false, + "options": [ + { + "label": "Explicit non-retryable list (recommended)", + "description": "Each of the 4 non-webhook workers declares its non-retryable error types (validation errors, auth/permission errors, malformed payload). Those bypass retry and land in the dead-letter set (R3) on attempt 1 with the error attached. Any error not on the list retries per R3/R4. Required proof: per worker, one test that a listed error goes to dead-letter without a retry, and one test that an unlisted error retries. Completeness 9/10. human: ~half day / CC: ~15 min." + }, + { + "label": "Retry everything to ceiling", + "description": "No classification. Every failure in the 4 non-webhook workers retries per R3/R4 until the ceiling, then lands in dead-letter. Required proof: covered by R3 tests. Completeness 6/10. human: 0 / CC: 0." + } + ] + } + ], + "answered": true, + "failed": false, + "answers": { + "D5 — Should workers name errors that must not be retried, or retry every failure to the ceiling?\nProject/branch/task: `main` of the plan fixture repo, plan \"Add background job retry framework\".\nELI10: Some failures fix themselves if you wait (a database hiccup, a slow API). Others never will (a payload that fails validation, a revoked API key). Retrying the second kind five times with growing delays just wastes capacity and delays the moment someone notices. Letting each worker say \"these error types are permanent\" sends them straight to the dead-letter set on the first try.\nStakes if we pick wrong: a bad payload burns 5 attempts and up to the full backoff window before it surfaces, and during a bad deploy every job does this at once.\nRecommendation: A because it is a small per-worker list, the dead-letter path already exists from D3, and it turns a permanent failure into an immediate signal instead of a delayed one. Medium confidence on which types are permanent; verify against the actual error classes when implementing.\nCompleteness: A=9/10, B=6/10\nPros / cons:\nA) Explicit non-retryable list (recommended)\n ✅ Permanent failures reach the dead-letter set on attempt 1, so operators see bad payloads or revoked credentials within seconds\n ✅ Unknown errors still default to retry, so nothing transient is accidentally dropped\n ❌ Each worker needs a short, reviewed list of permanent error types, and a wrong entry makes a transient error permanent\nB) Retry everything to ceiling\n ✅ No classification to get wrong; behavior is identical for every worker\n ✅ Nothing to maintain when new error types appear\n ❌ Permanent failures consume the full attempt budget and backoff window before anyone can see them\nNet: a short reviewed list per worker against a guaranteed delay on every permanent failure.": "Explicit non-retryable list (recommended)" + }, + "unansweredQuestionIndices": [], + "answeredAt": "2026-09-29T17:52:12.006Z" + }, + { + "sessionId": "cd04b55a-5f1a-4672-b3ed-40ea1b1636bc", + "toolUseId": "toolu_016w6Sj4WK8TpN3bzensgKcm", + "questions": [ + { + "question": "D6 — Extract one shared retry policy module now, or keep 5 copy-pasted envelopes?\nProject/branch/task: `main` of the plan fixture repo, plan \"Add background job retry framework\".\nELI10: You just decided the curve shape, jitter, the attempt ceiling, the dead-letter alert and how errors are classified. Each of those has to live somewhere. If the retry envelope stays copy-pasted in 5 workers, every one of those decisions is copied 5 times, and the first bug fix or tuning change has to be made in 5 places and tested 5 times. One small shared module applies each decision once and every worker gets it.\nStakes if we pick wrong: a curve or ceiling bug fixed in 4 of 5 workers, or a jitter change that lands in 3, and nobody notices until the fifth worker stampedes a downstream.\nRecommendation: A because the plan already states the 5 bodies are identical, the module is about 50 lines, the migration is mechanical per worker, and doing the refactor commit before the behavior commit keeps each step reviewable. (human: ~1 day / CC: ~30 min)\nCompleteness: A=9/10, B=6/10, C=3/10\nPros / cons:\nA) Extract now, migrate all 5 (recommended)\n ✅ Curve, jitter, ceiling, dead-letter alert and classifier shape are each implemented and tested exactly once\n ✅ Refactor commit lands before the behavior-change commit, so each is small and reviewable on its own\n ❌ A defect in the shared module affects all 5 workers at once; the contract tests are the guard\nB) Extract now, migrate webhook only\n ✅ Smallest first step; proves the module against the worker whose behavior is changing anyway\n ✅ Other 4 workers are untouched in this change, so their risk is zero for now\n ❌ Leaves 4 copies carrying the new policy by hand, so the duplication the plan already called out gets worse, not better\nC) Leave duplication\n ✅ No refactor risk in this change at all\n ✅ Matches the plan as written\n ❌ Every approved policy decision is copy-pasted 5 times and drifts from the first fix onward\nNet: one 50-line module now against 5 hand-maintained copies of every retry decision.", + "header": "Shared module", + "multiSelect": false, + "options": [ + { + "label": "Extract now, migrate all 5 (recommended)", + "description": "Create one `retryPolicy` module exporting backoffStrategy(attempt, rng) with equal jitter and a configurable max-delay clamp, DEFAULT_MAX_ATTEMPTS, onDeadLetter(job, err) emitting the structured log and metric, and isNonRetryable(err, list). Structured attempt log fields: job id, attempt, delay, error class, decision. Inline ASCII state diagram in the module header. All 5 workers register with the library through it, each passing its own non-retryable list. Land the refactor commit before the behavior-change commit. Required proof: shared-contract unit tests for each export plus one integration test per worker that the library invokes the shared policy on failure. Completeness 9/10. human: ~1 day / CC: ~30 min." + }, + { + "label": "Extract now, migrate webhook only", + "description": "Create the same `retryPolicy` module and migrate only the webhook worker in this change. The other 4 workers keep their copied envelopes and apply R3/R4/R5 by hand until a follow-up (TODO). Required proof: shared-contract unit tests plus one webhook integration test. Completeness 6/10. human: ~half day / CC: ~15 min." + }, + { + "label": "Leave duplication", + "description": "Keep 5 copy-pasted envelopes as the plan proposes. Apply R3/R4/R5 policy in each copy. No shared module, no shared tests; per-copy tests only. Completeness 3/10. human: ~1 day of copy-paste / CC: ~20 min." + } + ] + } + ], + "answered": true, + "failed": false, + "answers": { + "D6 — Extract one shared retry policy module now, or keep 5 copy-pasted envelopes?\nProject/branch/task: `main` of the plan fixture repo, plan \"Add background job retry framework\".\nELI10: You just decided the curve shape, jitter, the attempt ceiling, the dead-letter alert and how errors are classified. Each of those has to live somewhere. If the retry envelope stays copy-pasted in 5 workers, every one of those decisions is copied 5 times, and the first bug fix or tuning change has to be made in 5 places and tested 5 times. One small shared module applies each decision once and every worker gets it.\nStakes if we pick wrong: a curve or ceiling bug fixed in 4 of 5 workers, or a jitter change that lands in 3, and nobody notices until the fifth worker stampedes a downstream.\nRecommendation: A because the plan already states the 5 bodies are identical, the module is about 50 lines, the migration is mechanical per worker, and doing the refactor commit before the behavior commit keeps each step reviewable. (human: ~1 day / CC: ~30 min)\nCompleteness: A=9/10, B=6/10, C=3/10\nPros / cons:\nA) Extract now, migrate all 5 (recommended)\n ✅ Curve, jitter, ceiling, dead-letter alert and classifier shape are each implemented and tested exactly once\n ✅ Refactor commit lands before the behavior-change commit, so each is small and reviewable on its own\n ❌ A defect in the shared module affects all 5 workers at once; the contract tests are the guard\nB) Extract now, migrate webhook only\n ✅ Smallest first step; proves the module against the worker whose behavior is changing anyway\n ✅ Other 4 workers are untouched in this change, so their risk is zero for now\n ❌ Leaves 4 copies carrying the new policy by hand, so the duplication the plan already called out gets worse, not better\nC) Leave duplication\n ✅ No refactor risk in this change at all\n ✅ Matches the plan as written\n ❌ Every approved policy decision is copy-pasted 5 times and drifts from the first fix onward\nNet: one 50-line module now against 5 hand-maintained copies of every retry decision.": "Extract now, migrate all 5 (recommended)" + }, + "unansweredQuestionIndices": [], + "answeredAt": "2026-09-29T17:53:59.810Z" + }, + { + "sessionId": "cd04b55a-5f1a-4672-b3ed-40ea1b1636bc", + "toolUseId": "toolu_01Pv8bKkKS2rSamqKz1rvNX1", + "questions": [ + { + "question": "D7 — How do we protect the existing `processWebhookJob()` behavior through the rewrite?\nProject/branch/task: `main` of the plan fixture repo, plan \"Add background job retry framework\".\nELI10: You are rewriting the code that sends webhooks to customers, and there are no tests around it. The rewrite is supposed to keep everything the same except how failures are handled. Without tests written against the current code first, there is no way to know whether the new version still sends the same request, with the same headers and signature, exactly once. The cheapest insurance is to pin the current behavior in tests before changing a line, then keep them green.\nStakes if we pick wrong: a subtly different request body or signature ships to every webhook receiver at once, or a duplicate send slips through, and the first signal is a customer complaint.\nRecommendation: A because characterization tests are cheap with AI, they are the only way to detect an unintended difference in a rewrite, and they become the permanent contract suite for the webhook path. (human: ~1 day / CC: ~20 min)\nCompleteness: A=9/10, B=6/10\nPros / cons:\nA) Characterize first, then rewrite (recommended)\n ✅ Pins request shape, headers, signature and success bookkeeping against the current code, so any unintended difference fails a test\n ✅ Intentional changes (pre-send retry, dead-letter) are written as explicit tests, so the diff between old and new behavior is documented\n ❌ Requires reading the current implementation carefully and a day of test writing before the rewrite starts\nB) At-most-once assertions only\n ✅ Covers the one guarantee the plan named as at risk\n ✅ Faster to write; no characterization pass\n ❌ Request shape, headers, signature and success bookkeeping can change silently and no test notices\nNet: pin the whole current contract for a day of work, or protect one guarantee and hope the rest survived.", + "header": "Webhook regression", + "multiSelect": false, + "options": [ + { + "label": "Characterize first, then rewrite (recommended)", + "description": "Before modifying `processWebhookJob()`, write characterization tests (e.g. `processWebhookJob.test`) against the current implementation asserting: exact request body, headers and signature for a fixed payload; exactly one send on success, on timeout and on 5xx; success bookkeeping. Keep them green through the rewrite. Then add intentional-difference tests: pre-send failure schedules a retry; timeout/5xx lands in dead-letter with no second send. Flag the suite CRITICAL in the plan. R8 stays pending. Completeness 9/10. human: ~1 day / CC: ~20 min." + }, + { + "label": "At-most-once assertions only", + "description": "After the rewrite, add tests asserting exactly one send on success, timeout and 5xx, and a retry on pre-send failure. No characterization of request shape, headers, signature or success bookkeeping. R8 stays pending. Completeness 6/10. human: ~2 h / CC: ~5 min." + } + ] + } + ], + "answered": true, + "failed": false, + "answers": { + "D7 — How do we protect the existing `processWebhookJob()` behavior through the rewrite?\nProject/branch/task: `main` of the plan fixture repo, plan \"Add background job retry framework\".\nELI10: You are rewriting the code that sends webhooks to customers, and there are no tests around it. The rewrite is supposed to keep everything the same except how failures are handled. Without tests written against the current code first, there is no way to know whether the new version still sends the same request, with the same headers and signature, exactly once. The cheapest insurance is to pin the current behavior in tests before changing a line, then keep them green.\nStakes if we pick wrong: a subtly different request body or signature ships to every webhook receiver at once, or a duplicate send slips through, and the first signal is a customer complaint.\nRecommendation: A because characterization tests are cheap with AI, they are the only way to detect an unintended difference in a rewrite, and they become the permanent contract suite for the webhook path. (human: ~1 day / CC: ~20 min)\nCompleteness: A=9/10, B=6/10\nPros / cons:\nA) Characterize first, then rewrite (recommended)\n ✅ Pins request shape, headers, signature and success bookkeeping against the current code, so any unintended difference fails a test\n ✅ Intentional changes (pre-send retry, dead-letter) are written as explicit tests, so the diff between old and new behavior is documented\n ❌ Requires reading the current implementation carefully and a day of test writing before the rewrite starts\nB) At-most-once assertions only\n ✅ Covers the one guarantee the plan named as at risk\n ✅ Faster to write; no characterization pass\n ❌ Request shape, headers, signature and success bookkeeping can change silently and no test notices\nNet: pin the whole current contract for a day of work, or protect one guarantee and hope the rest survived.": "Characterize first, then rewrite (recommended)" + }, + "unansweredQuestionIndices": [], + "answeredAt": "2026-09-29T17:55:40.441Z" + }, + { + "sessionId": "cd04b55a-5f1a-4672-b3ed-40ea1b1636bc", + "toolUseId": "toolu_01N7xKjvDw5oJhSmyLK3zRAR", + "questions": [ + { + "question": "D8 — Do the per-worker integration tests run the real job library, or a mocked hook?\nProject/branch/task: `main` of the plan fixture repo, plan \"Add background job retry framework\".\nELI10: The whole point of D1 was that the library remembers attempt counts across worker restarts and keeps failed jobs somewhere you can find them. A test that fakes the library cannot check either of those; it only checks that your function was called. Running the real library against a throwaway test queue is slower but proves the parts you are relying on actually behave.\nStakes if we pick wrong: the retry count resets on every deploy or the failed set is empty when you need it, and every test was green because the mock said so.\nRecommendation: A because the library's persistence and failed set are load-bearing assumptions from D1 and D3, and a mock cannot verify either; the cost is a test backend fixture the library almost certainly already ships.\nCompleteness: A=9/10, B=6/10\nPros / cons:\nA) Real library backend in tests (recommended)\n ✅ Proves attempt count survives a worker restart and that exhausted jobs are actually in the failed set with their last error\n ✅ Catches library-version behavior changes and off-by-one attempt numbering that a stub would hide\n ❌ Slower suite and a test backend fixture to maintain (in-process queue or container)\nB) Mocked library hooks\n ✅ Fast, deterministic, no external fixture\n ✅ Enough to prove the worker wiring calls the shared policy\n ❌ Restart persistence and failed-set contents stay unverified, which are exactly the guarantees D1 and D3 depend on\nNet: a slower fixture that verifies the library promises you are betting on, or fast tests that trust them.", + "header": "Integration depth", + "multiSelect": false, + "options": [ + { + "label": "Real library backend in tests (recommended)", + "description": "Per-worker integration tests (5) run the actual job library against a test backend (in-process or containerized queue). Assertions: strategy invoked with real attempt numbers; attempt count survives a simulated worker restart mid-backoff; after the ceiling the job is in the failed set with its last error; a listed non-retryable error is in the failed set after attempt 1. Mark [E2E]. Completeness 9/10. human: ~1 day / CC: ~30 min." + }, + { + "label": "Mocked library hooks", + "description": "Per-worker tests stub the library retry hook and assert the shared policy is invoked with the expected arguments. No restart or failed-set verification. Completeness 6/10. human: ~2 h / CC: ~10 min." + } + ] + } + ], + "answered": true, + "failed": false, + "answers": { + "D8 — Do the per-worker integration tests run the real job library, or a mocked hook?\nProject/branch/task: `main` of the plan fixture repo, plan \"Add background job retry framework\".\nELI10: The whole point of D1 was that the library remembers attempt counts across worker restarts and keeps failed jobs somewhere you can find them. A test that fakes the library cannot check either of those; it only checks that your function was called. Running the real library against a throwaway test queue is slower but proves the parts you are relying on actually behave.\nStakes if we pick wrong: the retry count resets on every deploy or the failed set is empty when you need it, and every test was green because the mock said so.\nRecommendation: A because the library's persistence and failed set are load-bearing assumptions from D1 and D3, and a mock cannot verify either; the cost is a test backend fixture the library almost certainly already ships.\nCompleteness: A=9/10, B=6/10\nPros / cons:\nA) Real library backend in tests (recommended)\n ✅ Proves attempt count survives a worker restart and that exhausted jobs are actually in the failed set with their last error\n ✅ Catches library-version behavior changes and off-by-one attempt numbering that a stub would hide\n ❌ Slower suite and a test backend fixture to maintain (in-process queue or container)\nB) Mocked library hooks\n ✅ Fast, deterministic, no external fixture\n ✅ Enough to prove the worker wiring calls the shared policy\n ❌ Restart persistence and failed-set contents stay unverified, which are exactly the guarantees D1 and D3 depend on\nNet: a slower fixture that verifies the library promises you are betting on, or fast tests that trust them.": "Real library backend in tests (recommended)" + }, + "unansweredQuestionIndices": [], + "answeredAt": "2026-09-29T17:56:39.620Z" + }, + { + "sessionId": "cd04b55a-5f1a-4672-b3ed-40ea1b1636bc", + "toolUseId": "toolu_01XEdmtg1U25aqBrEGFb8dfg", + "questions": [ + { + "question": "D9 — Cache the dependency graph across retries, or recompute it on every attempt?\nProject/branch/task: `main` of the plan fixture repo, plan \"Add background job retry framework\".\nELI10: Every time a job retries, the plan reads the whole payload from the database again and rebuilds the same dependency graph from it. The payload never changes between attempts, so the answer is always the same. With up to 5 attempts, that is up to 5 reads and 5 builds per failing job, and failing jobs pile up exactly when something is already down. Building once and storing the result with the job removes almost all of that.\nStakes if we pick wrong: a downstream outage turns into a database load spike from your own retries, or you spend effort caching something that turns out to be cheap.\nRecommendation: A because the graph is derived from an immutable payload, the library already stores job data per attempt, and the guard (payload hash check) makes the cache safe; it also removes the redundant DB fetch. Confidence is medium: if payloads are tiny and the graph build is microseconds, C is acceptable and this becomes a TODO.\nCompleteness: A=9/10, B=5/10, C=4/10\nPros / cons:\nA) Compute once, store on job (recommended)\n ✅ Retries read no extra payload and build no graph; outage-time DB load drops from 5N to about N\n ✅ Survives worker restarts and works across workers because the cache lives in the job data, not in a process\n ❌ Adds serialized graph size to each job record and needs a payload-hash guard to stay correct\nB) In-process memo\n ✅ Simple to add, no change to job data shape\n ✅ Helps when the same worker process picks up the retry\n ❌ Retries usually land on a different worker or after a restart, so the memo misses most of the time and still re-fetches the payload\nC) Leave as-is\n ✅ Zero new code and no cache correctness to reason about\n ✅ Bounded at 5 attempts by D3, so the waste is finite\n ❌ Every retry storm during an outage multiplies database reads by up to 5\nNet: one persisted derived value with a hash guard, or accept a 5x read multiplier exactly when the system is least healthy.", + "header": "Graph caching", + "multiSelect": false, + "options": [ + { + "label": "Compute once, store on job (recommended)", + "description": "On attempt 1, read the payload (from the library's job data if it carries it, else one DB fetch), build the dependency graph, and persist the serialized graph plus a payload hash in the job data. On later attempts, verify the hash and deserialize; on mismatch, rebuild. Required proof: unit test that attempt 2+ performs no DB fetch and no graph build when the hash matches; test that a hash mismatch triggers a rebuild. Completeness 9/10. human: ~half day / CC: ~15 min." + }, + { + "label": "In-process memo", + "description": "Memoize the built graph per job id in worker memory. Payload re-fetch unchanged. Required proof: test that a second attempt in the same process reuses the graph. Completeness 5/10. human: ~1 h / CC: ~5 min." + }, + { + "label": "Leave as-is", + "description": "Re-fetch the payload and rebuild the graph on every attempt, as the plan proposes. No new tests. Completeness 4/10. human: 0 / CC: 0." + } + ] + } + ], + "answered": true, + "failed": false, + "answers": { + "D9 — Cache the dependency graph across retries, or recompute it on every attempt?\nProject/branch/task: `main` of the plan fixture repo, plan \"Add background job retry framework\".\nELI10: Every time a job retries, the plan reads the whole payload from the database again and rebuilds the same dependency graph from it. The payload never changes between attempts, so the answer is always the same. With up to 5 attempts, that is up to 5 reads and 5 builds per failing job, and failing jobs pile up exactly when something is already down. Building once and storing the result with the job removes almost all of that.\nStakes if we pick wrong: a downstream outage turns into a database load spike from your own retries, or you spend effort caching something that turns out to be cheap.\nRecommendation: A because the graph is derived from an immutable payload, the library already stores job data per attempt, and the guard (payload hash check) makes the cache safe; it also removes the redundant DB fetch. Confidence is medium: if payloads are tiny and the graph build is microseconds, C is acceptable and this becomes a TODO.\nCompleteness: A=9/10, B=5/10, C=4/10\nPros / cons:\nA) Compute once, store on job (recommended)\n ✅ Retries read no extra payload and build no graph; outage-time DB load drops from 5N to about N\n ✅ Survives worker restarts and works across workers because the cache lives in the job data, not in a process\n ❌ Adds serialized graph size to each job record and needs a payload-hash guard to stay correct\nB) In-process memo\n ✅ Simple to add, no change to job data shape\n ✅ Helps when the same worker process picks up the retry\n ❌ Retries usually land on a different worker or after a restart, so the memo misses most of the time and still re-fetches the payload\nC) Leave as-is\n ✅ Zero new code and no cache correctness to reason about\n ✅ Bounded at 5 attempts by D3, so the waste is finite\n ❌ Every retry storm during an outage multiplies database reads by up to 5\nNet: one persisted derived value with a hash guard, or accept a 5x read multiplier exactly when the system is least healthy.": "Compute once, store on job (recommended)" + }, + "unansweredQuestionIndices": [], + "answeredAt": "2026-09-29T17:58:34.047Z" + }, + { + "sessionId": "cd04b55a-5f1a-4672-b3ed-40ea1b1636bc", + "toolUseId": "toolu_012XHAmgKwXGVbhDkoCK1K1R", + "questions": [ + { + "question": "D10 — Record the webhook at-least-once upgrade as a TODO?\nProject/branch/task: gstack-plan-count-yWJb6k on main, retry framework plan; follow-up to D2.\nELI10: In D2 you chose to keep webhooks at \"send at most once\", so a slow or erroring receiver means that delivery is dropped into the failed set instead of retried. The fix (retry with a delivery id the receiver can dedupe) needs receiver work first. This question only decides whether we write that follow-up down in TODOS.md so it does not get lost.\nStakes if we pick wrong: skipped, the only trace is a code comment and a decision-log row; built now, this PR grows and contradicts the D2 call.\nRecommendation: A because the upgrade has an external prerequisite and a clear trigger, which is exactly what a TODO is for.\nNote: options differ in kind, not coverage — no completeness score.\nPros / cons:\nA) Add to TODOS.md (recommended)\n ✅ Creates TODOS.md with the What/Why/Context/Depends-on record, findable by /retro and future reviews\n ✅ Zero implementation cost now; the marker in code and the TODO entry point at each other\n ❌ One more file in the repo that someone has to keep honest as work lands\nB) Skip\n ✅ No new file; the decision log and gstack-shortcut marker already carry the trigger\n ✅ Avoids a TODO nobody may pick up if receivers never add dedupe\n ❌ The trigger lives only in a comment and a JSONL row, easy to miss when receivers do change\nC) Build it now in this PR\n ✅ Ships the stronger delivery guarantee in the same change as the retry framework\n ✅ Reuses the retryPolicy module while it is fresh\n ❌ Reverses D2 and depends on receiver dedupe that does not exist yet, so duplicates would reach receivers\nNet: a TODO entry now versus relying on a code marker alone; building now is off the table until receivers can dedupe.", + "header": "Webhook TODO", + "multiSelect": false, + "options": [ + { + "label": "Add to TODOS.md (recommended)", + "description": "Create TODOS.md at implementation time with the TODO record (What/Why/Context/Depends-on) under a `## Workers` section, P3, effort M. No product code change." + }, + { + "label": "Skip", + "description": "Do not create TODOS.md. The decision log entry and the gstack-shortcut marker remain the only trail." + }, + { + "label": "Build it now in this PR", + "description": "Extend the accepted scope to at-least-once delivery with delivery id and idempotency key; would reopen D2." + } + ] + } + ], + "answered": true, + "failed": false, + "answers": { + "D10 — Record the webhook at-least-once upgrade as a TODO?\nProject/branch/task: gstack-plan-count-yWJb6k on main, retry framework plan; follow-up to D2.\nELI10: In D2 you chose to keep webhooks at \"send at most once\", so a slow or erroring receiver means that delivery is dropped into the failed set instead of retried. The fix (retry with a delivery id the receiver can dedupe) needs receiver work first. This question only decides whether we write that follow-up down in TODOS.md so it does not get lost.\nStakes if we pick wrong: skipped, the only trace is a code comment and a decision-log row; built now, this PR grows and contradicts the D2 call.\nRecommendation: A because the upgrade has an external prerequisite and a clear trigger, which is exactly what a TODO is for.\nNote: options differ in kind, not coverage — no completeness score.\nPros / cons:\nA) Add to TODOS.md (recommended)\n ✅ Creates TODOS.md with the What/Why/Context/Depends-on record, findable by /retro and future reviews\n ✅ Zero implementation cost now; the marker in code and the TODO entry point at each other\n ❌ One more file in the repo that someone has to keep honest as work lands\nB) Skip\n ✅ No new file; the decision log and gstack-shortcut marker already carry the trigger\n ✅ Avoids a TODO nobody may pick up if receivers never add dedupe\n ❌ The trigger lives only in a comment and a JSONL row, easy to miss when receivers do change\nC) Build it now in this PR\n ✅ Ships the stronger delivery guarantee in the same change as the retry framework\n ✅ Reuses the retryPolicy module while it is fresh\n ❌ Reverses D2 and depends on receiver dedupe that does not exist yet, so duplicates would reach receivers\nNet: a TODO entry now versus relying on a code marker alone; building now is off the table until receivers can dedupe.": "Add to TODOS.md (recommended)" + }, + "unansweredQuestionIndices": [], + "answeredAt": "2026-09-29T18:01:55.194Z" + } + ], + "plan": "# Engineering review: Add background job retry framework\n\nReview target (fixed): `PLAN.md` in `/home/runner/.cache/gstack-paid-shard-mmiSh1/tmp/gstack-plan-count-yWJb6k` (branch `main`, commit `0ad2479`).\nReport file: this file (path requested by the user).\nReviewer: /plan-eng-review, session `642-1790703811-6005ed0c`, 2026-09-29.\n\n## Original plan (unchanged copy)\n\n# Plan: Add background job retry framework\n\n## Architecture\nWe'll roll a custom exponential-backoff scheduler inline in each worker\nrather than use the existing job library's built-in retry hooks. Same\nshape as the library version, but we want full control over the curve.\n\n## Code quality\nThe retry envelope (compute delay, log attempt, dispatch) is duplicated\nacross 5 worker files with copy-pasted bodies. We will leave the\nduplication for now and refactor \"later.\"\n\n## Tests\nThe existing `processWebhookJob()` flow gets rewritten as part of this\nchange. No regression test for the prior at-most-once delivery guarantee\nis planned.\n\n## Performance\nOn every retry we re-fetch the full job payload from the database, then\niterate the payload to recompute the dependency graph. Could cache the\ngraph on the first attempt; not planned.\n\n## Scope Challenge record\n\nEvidence available: plan text only. The repo contains `PLAN.md` and `CLAUDE.md`; the 5 worker files, `processWebhookJob()`, the job library and its retry hooks are `not available` in this checkout. Findings quote plan lines and are calibrated as plan-text findings.\n\nComplexity count (estimates from plan text): ~5-6 changed files (5 worker files; `processWebhookJob()` may live in one of them), 0 new classes/services (scheduler is inline). Below the 8-file / 2-class gate, so the complexity selectors (B) are skipped.\n\nSearch check: Aside unavailable, host WebSearch used. Industry default [Layer 1]: library built-in retry, exponential backoff + jitter, bounded attempts, dead-letter, idempotent handlers.\n\n## Decision ledger\n\n### R1: Retry scheduler mechanism (library hooks vs custom inline scheduler)\nFinding: SC-1, P1, confidence 8/10, PLAN.md:7-9, reviewer: plan-eng-review (native)\nPlan baseline: original proposal, \"custom exponential-backoff scheduler inline in each worker rather than use the existing job library's built-in retry hooks\" (PLAN.md:7-9). Nothing approved yet.\nRuntime evidence: unknown. Job library and worker files not available in this checkout; plan text states the library has built-in retry hooks and the custom version is the \"same shape\".\nComparison grid:\n\n| Choice | Current | A) Library hooks + custom curve | B) Custom inline scheduler |\n|---|---|---|---|\n| R1 retry mechanism | custom inline scheduler (proposed) | library retry hooks, backoff supplied as one strategy function | custom scheduler inline per worker, as proposed |\n| Backoff curve ownership | \"full control\" wanted | full control via strategy function (verify hook accepts a function; else fall back to B) | full control |\n| Attempt count persistence / terminal handling | unspecified | inherited from library | must be hand-built (pending, R3) |\n| R2 webhook delivery semantics | pending | pending | pending |\n| R3 attempt bound + dead-letter | pending | pending | pending |\n| R4 jitter | pending | pending | pending |\n| R5 shared envelope | pending | pending | pending |\n\nQuestion D1:\nD1 — Use the job library's retry hooks or roll a custom inline scheduler?\nProject/branch/task: `main` of the plan fixture repo, plan \"Add background job retry framework\".\nELI10: The job library you already run has a retry system built in. The plan wants to rebuild the same thing by hand inside each of the 5 workers, only so the delay curve can be tuned. Retry systems are easy to get subtly wrong: losing the attempt count when a worker restarts, retrying forever, or retrying twice at once. The library has already paid for those bugs; a hand-rolled copy pays for them again in production.\nStakes if we pick wrong: a hand-rolled scheduler that drops attempt state on restart or double-schedules turns one failed job into duplicate side effects or an infinite retry loop, with no dead-letter to catch it.\nRecommendation: A because the plan admits the shapes are identical, and the curve is pluggable in the library through a strategy function, so A delivers the same control with far less new code. (human: ~1 day / CC: ~20 min for A; human: ~1 week / CC: ~2 h for B plus ongoing ownership)\nCompleteness: A=9/10, B=5/10\nPros / cons:\nA) Library hooks + custom curve (recommended)\n ✅ Attempt counting, persistence across restarts and terminal handling come from tested library code, not new code\n ✅ The custom curve still lives in one strategy function, so \"full control over the curve\" is preserved\n ❌ Requires confirming the library's hook accepts a custom delay function; if it does not, we fall back to B for the curve only\nB) Custom inline scheduler\n ✅ Zero dependency on the library's retry semantics or its upgrade cadence\n ✅ Any curve shape, any bookkeeping, no hook constraints\n ❌ Rebuilds attempt state, restart persistence, concurrency guards and dead-lettering by hand, and those are the parts that fail at 3am\nNet: you are trading a one-line strategy function against owning a second retry engine forever.\nHeader: Retry engine\nOptions:\nA) Library hooks + custom curve (recommended)\nRegister the exponential-backoff curve as one custom backoff strategy function with the job library's built-in retry hooks. Attempt counting, persistence across worker restarts, and terminal/dead-letter handling come from the library. Verify the hook accepts a delay function first; if it does not, fall back to a custom curve only for delay computation while keeping library scheduling. R2-R5 stay pending. Completeness 9/10. human: ~1 day / CC: ~20 min.\nB) Custom inline scheduler\nKeep the plan as written: a custom exponential-backoff scheduler inline in each worker, bypassing the library's retry hooks. Attempt state, restart persistence, concurrency guards and terminal handling must be designed and tested by hand (tracked under R3). R2-R5 stay pending. Completeness 5/10. human: ~1 week / CC: ~2 h plus ongoing ownership.\n\nState: approved\nActual answer: A) Library hooks + custom curve (D1 answer, user selection)\nAccepted scope: Replace the custom inline scheduler with the job library's built-in retry hooks. The exponential-backoff curve is supplied as one custom backoff strategy function. Attempt counting, persistence across worker restarts and terminal/dead-letter handling come from the library. First implementation step: verify the hook accepts a delay function; if it does not, use a custom delay computation only, keeping library scheduling. Required proof: unit tests of the strategy function (curve values per attempt) and an integration test that the library invokes it on failure. R2-R5 remain pending.\nHistory: none\n\nScope Challenge result: scope accepted as-is (D1 changed mechanism, not feature scope). MODE = FULL_REVIEW.\n\n### R2: Webhook delivery semantics under retry\nFinding: ARCH-1, P1, confidence 8/10, PLAN.md:17-19, reviewer: plan-eng-review (native)\nPlan baseline: original proposal, `processWebhookJob()` is rewritten to retry; prior guarantee was at-most-once; no idempotency key or retry classification stated (PLAN.md:17-19). R1 approved: retries run through library hooks.\nRuntime evidence: unknown. `processWebhookJob()` and receiver contract not available in this checkout. Plan text asserts the prior guarantee was at-most-once.\nComparison grid:\n\n| Choice | Current | A) Keep at-most-once | B) At-least-once + idempotency key | C) Plain retry (plan as written) |\n|---|---|---|---|---|\n| R2 webhook delivery semantics | at-most-once today; plan retries without stating semantics | at-most-once preserved: retry only when the request provably never left (connect/DNS/pre-send errors); timeouts and 5xx are terminal | at-least-once: retry timeouts/5xx too; every attempt carries the same stable delivery id header so receivers can dedupe | at-least-once with duplicates indistinguishable to receivers |\n| Receiver-visible contract | no duplicates | no duplicates (unchanged) | duplicates possible, always carrying the same id (contract change, communicate to receivers) | duplicates possible, not deduplicable |\n| R1 library hooks | approved | approved, unchanged | approved, unchanged | approved, unchanged |\n| R3 attempt bound + dead-letter | pending | pending | pending | pending |\n| R4 jitter | pending | pending | pending | pending |\n| R5 error classification (other workers) | pending | pending (webhook classification fixed by this row) | pending (webhook classification fixed by this row) | pending |\n| R7 regression contract | pending | pending | pending | pending |\n\nQuestion D2:\nD2 — What delivery guarantee does `processWebhookJob()` keep once it retries?\nProject/branch/task: `main` of the plan fixture repo, plan \"Add background job retry framework\".\nELI10: Today a webhook is sent once; if it fails, it is gone (at-most-once). The moment you retry, a request that timed out after the customer already received it gets sent again, so the customer sees the same event twice. You have to pick: either only retry when you are sure the request never left, or retry freely but stamp every attempt with the same id so the customer can ignore repeats. The plan does neither and just retries.\nStakes if we pick wrong: customers process duplicate events (double orders, double emails) with no way to detect them, or you ship a retry feature that almost never fires because most webhook failures are timeouts.\nRecommendation: B because it is the standard webhook contract (retry on timeout/5xx, stable event id per attempt) and is the only option where retrying actually improves delivery while giving receivers a way to dedupe. This is a receiver-visible contract change; A is the right pick if you cannot communicate it to receivers.\nCompleteness: A=7/10, B=9/10, C=3/10\nPros / cons:\nA) Keep at-most-once\n ✅ No change to what receivers see; the existing guarantee and its regression test stay valid as-is\n ✅ Smallest blast radius: no new headers, no receiver communication needed\n ❌ Retries only fire on connect/DNS/pre-send errors; timeouts and 5xx go straight to terminal, so most real failures are still not retried\nB) At-least-once + idempotency key (recommended)\n ✅ Timeouts and 5xx are retried, so delivery reliability actually improves for receivers\n ✅ Same delivery id on every attempt lets receivers dedupe; this is the contract Stripe/GitHub-style webhooks use\n ❌ Receiver-visible contract change: duplicates become possible and receivers must be told to dedupe on the id\nC) Plain retry (plan as written)\n ✅ Least code: no header, no classification, just retry on any failure\n ✅ Ships fastest\n ❌ Duplicates reach receivers with no way to tell them apart; silent double side effects\nNet: A keeps the promise but retries little; B changes the promise but makes retries worth having; C breaks the promise silently.\nHeader: Webhook semantics\nOptions:\nA) Keep at-most-once\nPreserve at-most-once for `processWebhookJob()`. Retry only failures where the request provably never left the process (connection refused, DNS failure, errors before send). Timeouts, 5xx and any post-send ambiguity are terminal and go to whatever terminal handling R3 decides. No new headers; receiver contract unchanged. R3, R4, R5, R7 stay pending. Completeness 7/10. human: ~half day / CC: ~10 min.\nB) At-least-once + idempotency key (recommended)\nRetry timeouts and 5xx as well as pre-send failures. Every attempt carries the same stable delivery id (e.g. an `Idempotency-Key`/event-id header derived from the job id, generated once, persisted with the job). 4xx other than 408/429 is non-retryable for webhooks. Document to receivers that duplicates are possible and must be deduped on the id. R3, R4, R5, R7 stay pending. Completeness 9/10. human: ~1 day / CC: ~20 min.\nC) Plain retry (plan as written)\nRetry `processWebhookJob()` on any failure with no delivery id and no classification. Duplicates reach receivers undetectably. R3, R4, R5, R7 stay pending. Completeness 3/10. human: ~1 h / CC: ~5 min.\n\nState: approved\nActual answer: A) Keep at-most-once (D2 answer, user selection)\nAccepted scope: `processWebhookJob()` preserves at-most-once delivery. Retry fires only for failures where the request provably never left the process (connection refused, DNS failure, errors raised before send). Timeouts, 5xx responses and any post-send ambiguity are terminal and route to the terminal handling decided in R3. No new headers; receiver contract unchanged. Accepted shortcut (Completeness 7/10): ceiling is that timeouts/5xx are never retried; upgrade trigger is when receivers can dedupe on a stable delivery id, at which point revisit toward at-least-once + idempotency key. Required proof: regression test that a timeout/5xx produces exactly one send and no retry; test that a pre-send failure retries. R3, R4, R5, R7 remain pending. Decision log id: f9e8dfdf-4e90-4883-9064-014d784b9405.\nHistory: none\n\n### R3: Attempt ceiling and terminal handling (dead-letter)\nFinding: ARCH-2, P1, confidence 8/10, PLAN.md:7-9, reviewer: plan-eng-review (native)\nPlan baseline: original proposal names an exponential-backoff curve with no maximum attempts and no behavior on exhaustion (PLAN.md:7-9). R1 approved: library hooks. R2 approved: webhook timeouts/5xx are terminal and route to this row's handling.\nRuntime evidence: unknown. Library's dead-letter/failed-set feature not verifiable in this checkout.\nComparison grid:\n\n| Choice | Current | A) Bounded + dead-letter + alert | B) Bounded + log-and-drop | C) Unbounded (plan as written) |\n|---|---|---|---|---|\n| R3 attempt ceiling | none stated | max attempts per worker, default 5, configured in one place | max attempts per worker, default 5 | no ceiling |\n| R3 terminal disposition | none stated | exhausted and non-retryable jobs land in the library's dead-letter/failed set with last error; one structured error log + metric on entry | error log only, job dropped | never terminal (retries forever) |\n| R1 library hooks | approved | approved, unchanged | approved, unchanged | approved, unchanged |\n| R2 webhook at-most-once | approved | approved; webhook timeouts/5xx land in dead-letter | approved; webhook timeouts/5xx logged and dropped | approved (conflict: terminal has no destination) |\n| R4 jitter | pending | pending | pending | pending |\n| R5 error classification | pending | pending | pending | pending |\n\nQuestion D3:\nD3 — How many times may a job retry, and where does it go when it gives up?\nProject/branch/task: `main` of the plan fixture repo, plan \"Add background job retry framework\".\nELI10: A retry curve without a stopping point is a job that runs forever when the thing it depends on is down for good. You need a maximum number of tries, and you need a place for jobs that used up their tries (a dead-letter set) so someone can look at them and replay them later. Otherwise failed work quietly disappears or quietly never stops.\nStakes if we pick wrong: either a poisoned job hammers a downstream forever and starves healthy jobs, or failed webhooks and jobs vanish with only a log line nobody reads.\nRecommendation: A because the library already provides the failed set, so the dead-letter and alert cost a config line and one log call, and it is the only option where an operator can find and replay a lost job.\nCompleteness: A=9/10, B=6/10, C=2/10\nPros / cons:\nA) Bounded + dead-letter + alert (recommended)\n ✅ Exhausted jobs are inspectable and replayable from the library's failed set, with the last error attached\n ✅ One structured log line plus a metric on dead-letter entry makes a downstream outage visible within minutes\n ❌ Needs a per-worker ceiling value and a dead-letter retention/cleanup policy to be chosen and documented\nB) Bounded + log-and-drop\n ✅ Bounds the retry loop with the least configuration\n ✅ No dead-letter retention to manage\n ❌ A dropped job is gone; the only trace is a log line, so replay after an outage is impossible\nC) Unbounded (plan as written)\n ✅ No ceiling to tune; a job eventually succeeds if the dependency ever recovers\n ✅ Zero extra code\n ❌ Permanently failing jobs retry forever, consume worker capacity and never surface as a problem\nNet: you are choosing whether a job that cannot succeed becomes a visible artifact, a log line, or a permanent background load.\nHeader: Attempt ceiling\nOptions:\nA) Bounded + dead-letter + alert (recommended)\nSet a maximum attempt count per worker (default 5, overridable per worker, configured in the same place as the backoff strategy). On exhaustion or on a non-retryable error, the job lands in the library's dead-letter/failed set with its last error; emit one structured error log and a metric on entry. Webhook timeouts/5xx (terminal per R2) land here too. Document the retention/replay procedure. R4, R5 stay pending. Completeness 9/10. human: ~half day / CC: ~15 min.\nB) Bounded + log-and-drop\nSet the same per-worker maximum attempt count (default 5). On exhaustion, log the error and drop the job; no dead-letter set, no metric, no replay. R4, R5 stay pending. Completeness 6/10. human: ~1 h / CC: ~5 min.\nC) Unbounded (plan as written)\nNo attempt ceiling; the exponential curve continues until the job succeeds. No terminal handling. Conflicts with R2, which needs a terminal destination for webhook timeouts. R4, R5 stay pending. Completeness 2/10. human: 0 / CC: 0.\n\nState: approved\nActual answer: A) Bounded + dead-letter + alert (D3 answer, user selection)\nAccepted scope: Maximum attempt count per worker, default 5, overridable per worker, configured in the same place as the backoff strategy. On exhaustion or on a non-retryable error the job lands in the library's dead-letter/failed set with its last error; one structured error log and one metric are emitted on entry. Webhook timeouts/5xx (terminal per R2) land here too. Retention/replay procedure documented. Required proof: test that attempt N+1 is never scheduled after the ceiling; test that an exhausted job appears in the failed set with its last error and that the log/metric fire once; test that a webhook timeout lands in the failed set on attempt 1. R4, R5 remain pending.\nHistory: none\n\n### R4: Jitter on the backoff curve\nFinding: ARCH-3, P2, confidence 7/10, PLAN.md:7, reviewer: plan-eng-review (native)\nPlan baseline: original proposal, \"custom exponential-backoff scheduler\" with \"full control over the curve\" (PLAN.md:7-9); no jitter mentioned. R1 approved: curve lives in one strategy function.\nRuntime evidence: unknown. No curve code available.\nComparison grid:\n\n| Choice | Current | A) Equal jitter | B) No jitter (pure curve) |\n|---|---|---|---|\n| R4 jitter | unspecified (pure `base * 2^attempt` implied) | delay = half of the curve value plus a random amount up to the other half, so retries spread across the window | delay = exact curve value; all jobs failing at time T retry at T+delay together |\n| Curve ownership (R1) | approved: one strategy function | unchanged; jitter applied inside the same function | unchanged |\n| Delay ceiling | unspecified | unspecified (implicitly bounded by R3 max attempts) | unspecified |\n| R3 attempt ceiling | approved | approved, unchanged | approved, unchanged |\n| R5 error classification | pending | pending | pending |\n\nQuestion D4:\nD4 — Add jitter to the backoff curve, or keep it deterministic?\nProject/branch/task: `main` of the plan fixture repo, plan \"Add background job retry framework\".\nELI10: When a downstream service goes down, hundreds of jobs across all 5 workers fail at the same moment. With a pure exponential curve they all wake up at exactly the same moment too, and hit the recovering service as one wave, which can knock it over again. Jitter adds a random spread to each delay so the retries trickle back instead of stampeding.\nStakes if we pick wrong: a downstream that recovers from an outage gets re-flattened by your own synchronized retry wave, turning a 2-minute blip into a 20-minute incident.\nRecommendation: A because it is two lines inside the strategy function you already own and it is the standard mitigation for retry storms; deterministic curves are only useful in tests, which can seed or stub the random source.\nCompleteness: A=9/10, B=6/10\nPros / cons:\nA) Equal jitter (recommended)\n ✅ Retries after a shared outage spread across the window instead of returning as one synchronized burst\n ✅ Lives inside the single strategy function from D1, so every worker gets it with no per-worker code\n ❌ Curve tests need an injectable random source to stay deterministic\nB) No jitter (pure curve)\n ✅ Exact, predictable retry times that are easy to reason about and assert in tests\n ✅ Zero extra code beyond the curve itself\n ❌ All jobs that fail together retry together, so the retry framework itself becomes a traffic amplifier during outages\nNet: predictability in tests against stampede protection in production; the test cost is one injected random source.\nHeader: Jitter\nOptions:\nA) Equal jitter (recommended)\nInside the single backoff strategy function, compute the exponential delay and return half of it plus a random amount up to the other half (equal jitter). The random source is injectable so tests can pin it. Required proof: unit test that returned delays fall within [curve/2, curve] for each attempt, and that a pinned random source gives a deterministic value. R5 stays pending. Completeness 9/10. human: ~1 h / CC: ~5 min.\nB) No jitter (pure curve)\nReturn the exact exponential curve value with no random spread. Required proof: unit test of the exact value per attempt. R5 stays pending. Completeness 6/10. human: ~30 min / CC: ~3 min.\n\nState: approved\nActual answer: A) Equal jitter (D4 answer, user selection)\nAccepted scope: The single backoff strategy function computes the exponential delay and returns half of it plus a random amount up to the other half (equal jitter). Random source is injectable. Required proof: unit test that returned delays fall within [curve/2, curve] for each attempt; unit test that a pinned random source yields a deterministic value. R5 remains pending.\nHistory: none\n\n### R5: Retryable vs non-retryable error classification (non-webhook workers)\nFinding: ARCH-4, P2, confidence 6/10 (medium: actual error types not available), PLAN.md:7-9, reviewer: plan-eng-review (native)\nPlan baseline: original proposal retries on failure with no classification (PLAN.md:7-9). R2 fixed the webhook worker's classification (pre-send only). R3 approved: non-retryable errors route to dead-letter.\nRuntime evidence: unknown. Worker error types not available in this checkout.\nComparison grid:\n\n| Choice | Current | A) Explicit non-retryable list | B) Retry everything to ceiling |\n|---|---|---|---|\n| R5 classification | none; every failure retries | each worker declares its non-retryable error types (validation, auth/permission, malformed payload); those go straight to dead-letter; unknown errors retry | every error retries until the R3 ceiling, then dead-letter |\n| R2 webhook classification | approved (pre-send only) | unchanged | unchanged |\n| R3 ceiling + dead-letter | approved | unchanged; non-retryable short-circuits to dead-letter on attempt 1 | unchanged |\n| R4 jitter | approved | unchanged | unchanged |\n\nQuestion D5:\nD5 — Should workers name errors that must not be retried, or retry every failure to the ceiling?\nProject/branch/task: `main` of the plan fixture repo, plan \"Add background job retry framework\".\nELI10: Some failures fix themselves if you wait (a database hiccup, a slow API). Others never will (a payload that fails validation, a revoked API key). Retrying the second kind five times with growing delays just wastes capacity and delays the moment someone notices. Letting each worker say \"these error types are permanent\" sends them straight to the dead-letter set on the first try.\nStakes if we pick wrong: a bad payload burns 5 attempts and up to the full backoff window before it surfaces, and during a bad deploy every job does this at once.\nRecommendation: A because it is a small per-worker list, the dead-letter path already exists from D3, and it turns a permanent failure into an immediate signal instead of a delayed one. Medium confidence on which types are permanent; verify against the actual error classes when implementing.\nCompleteness: A=9/10, B=6/10\nPros / cons:\nA) Explicit non-retryable list (recommended)\n ✅ Permanent failures reach the dead-letter set on attempt 1, so operators see bad payloads or revoked credentials within seconds\n ✅ Unknown errors still default to retry, so nothing transient is accidentally dropped\n ❌ Each worker needs a short, reviewed list of permanent error types, and a wrong entry makes a transient error permanent\nB) Retry everything to ceiling\n ✅ No classification to get wrong; behavior is identical for every worker\n ✅ Nothing to maintain when new error types appear\n ❌ Permanent failures consume the full attempt budget and backoff window before anyone can see them\nNet: a short reviewed list per worker against a guaranteed delay on every permanent failure.\nHeader: Error classes\nOptions:\nA) Explicit non-retryable list (recommended)\nEach of the 4 non-webhook workers declares its non-retryable error types (validation errors, auth/permission errors, malformed payload). Those bypass retry and land in the dead-letter set (R3) on attempt 1 with the error attached. Any error not on the list retries per R3/R4. Required proof: per worker, one test that a listed error goes to dead-letter without a retry, and one test that an unlisted error retries. Completeness 9/10. human: ~half day / CC: ~15 min.\nB) Retry everything to ceiling\nNo classification. Every failure in the 4 non-webhook workers retries per R3/R4 until the ceiling, then lands in dead-letter. Required proof: covered by R3 tests. Completeness 6/10. human: 0 / CC: 0.\n\nState: approved\nActual answer: A) Explicit non-retryable list (D5 answer, user selection)\nAccepted scope: Each of the 4 non-webhook workers declares its non-retryable error types (validation, auth/permission, malformed payload; verify against actual error classes at implementation). Listed errors bypass retry and land in the dead-letter set (R3) on attempt 1 with the error attached. Unlisted errors retry per R3/R4. Required proof: per worker, one test that a listed error goes to dead-letter without a retry and one test that an unlisted error retries.\nHistory: none\n\nSection 1 dispositions: ARCH-1 (R2) accepted as at-most-once preserved; ARCH-2 (R3) accepted; ARCH-3 (R4) accepted; ARCH-4 (R5) accepted. Suppressed: webhook signature timestamp on retried attempts (confidence 4).\n\n### R6: Shared retry policy module vs duplicated envelope\nFinding: CQ-1, P1, confidence 9/10, PLAN.md:12-14, reviewer: plan-eng-review (native). Also carries CQ-2 (structured attempt log contract, P2, 7/10) and CQ-3 (delay clamp edge case, P2, 7/10) as contract details of the shared module.\nPlan baseline: original proposal, \"duplicated across 5 worker files with copy-pasted bodies. We will leave the duplication for now and refactor later\" (PLAN.md:12-14). R1, R3, R4, R5 approved: curve, ceiling, dead-letter, classification are now policy that each worker must apply.\nRuntime evidence: unknown. Worker files not available; plan asserts identical bodies.\nShared-code rubric: 5 proposed callers (plan assumption); identical behavior stated by plan; helper = one `retryPolicy` module (backoffStrategy, DEFAULT_MAX_ATTEMPTS, onDeadLetter, isNonRetryable); est. implementation removed 100-150, added 55-75, saved 45-95; tests add 80-120 so total diff may grow; blast radius all 5 workers, mitigated by contract tests and library scheduling.\nComparison grid:\n\n| Choice | Current | A) Extract now, migrate all 5 | B) Extract now, migrate webhook only | C) Leave duplication |\n|---|---|---|---|---|\n| R6 shared module | none; 5 copies | one `retryPolicy` module; all 5 workers register through it in this change, refactor commit before behavior commit | one `retryPolicy` module; webhook worker migrated now, other 4 keep copies until a follow-up | 5 copies of curve/ceiling/dead-letter/classifier config |\n| Structured attempt log (CQ-2) | unspecified | in shared module: job id, attempt, delay, error class, decision | in shared module, webhook only | per copy, unspecified |\n| Delay clamp (CQ-3) | unspecified | in shared strategy: clamp at configurable max delay | in shared strategy, webhook only | per copy, unspecified |\n| Inline ASCII state diagram | none | in shared module header | in shared module header | none |\n| R1/R3/R4/R5 approved policy | approved | applied once | applied once for webhook, 4 copies otherwise | applied 5 times |\n\nQuestion D6:\nD6 — Extract one shared retry policy module now, or keep 5 copy-pasted envelopes?\nProject/branch/task: `main` of the plan fixture repo, plan \"Add background job retry framework\".\nELI10: You just decided the curve shape, jitter, the attempt ceiling, the dead-letter alert and how errors are classified. Each of those has to live somewhere. If the retry envelope stays copy-pasted in 5 workers, every one of those decisions is copied 5 times, and the first bug fix or tuning change has to be made in 5 places and tested 5 times. One small shared module applies each decision once and every worker gets it.\nStakes if we pick wrong: a curve or ceiling bug fixed in 4 of 5 workers, or a jitter change that lands in 3, and nobody notices until the fifth worker stampedes a downstream.\nRecommendation: A because the plan already states the 5 bodies are identical, the module is about 50 lines, the migration is mechanical per worker, and doing the refactor commit before the behavior commit keeps each step reviewable. (human: ~1 day / CC: ~30 min)\nCompleteness: A=9/10, B=6/10, C=3/10\nPros / cons:\nA) Extract now, migrate all 5 (recommended)\n ✅ Curve, jitter, ceiling, dead-letter alert and classifier shape are each implemented and tested exactly once\n ✅ Refactor commit lands before the behavior-change commit, so each is small and reviewable on its own\n ❌ A defect in the shared module affects all 5 workers at once; the contract tests are the guard\nB) Extract now, migrate webhook only\n ✅ Smallest first step; proves the module against the worker whose behavior is changing anyway\n ✅ Other 4 workers are untouched in this change, so their risk is zero for now\n ❌ Leaves 4 copies carrying the new policy by hand, so the duplication the plan already called out gets worse, not better\nC) Leave duplication\n ✅ No refactor risk in this change at all\n ✅ Matches the plan as written\n ❌ Every approved policy decision is copy-pasted 5 times and drifts from the first fix onward\nNet: one 50-line module now against 5 hand-maintained copies of every retry decision.\nHeader: Shared module\nOptions:\nA) Extract now, migrate all 5 (recommended)\nCreate one `retryPolicy` module exporting backoffStrategy(attempt, rng) with equal jitter and a configurable max-delay clamp, DEFAULT_MAX_ATTEMPTS, onDeadLetter(job, err) emitting the structured log and metric, and isNonRetryable(err, list). Structured attempt log fields: job id, attempt, delay, error class, decision. Inline ASCII state diagram in the module header. All 5 workers register with the library through it, each passing its own non-retryable list. Land the refactor commit before the behavior-change commit. Required proof: shared-contract unit tests for each export plus one integration test per worker that the library invokes the shared policy on failure. Completeness 9/10. human: ~1 day / CC: ~30 min.\nB) Extract now, migrate webhook only\nCreate the same `retryPolicy` module and migrate only the webhook worker in this change. The other 4 workers keep their copied envelopes and apply R3/R4/R5 by hand until a follow-up (TODO). Required proof: shared-contract unit tests plus one webhook integration test. Completeness 6/10. human: ~half day / CC: ~15 min.\nC) Leave duplication\nKeep 5 copy-pasted envelopes as the plan proposes. Apply R3/R4/R5 policy in each copy. No shared module, no shared tests; per-copy tests only. Completeness 3/10. human: ~1 day of copy-paste / CC: ~20 min.\n\nState: approved\nActual answer: A) Extract now, migrate all 5 (D6 answer, user selection)\nAccepted scope: One `retryPolicy` module exporting backoffStrategy(attempt, rng) (exponential, equal jitter per R4, configurable max-delay clamp), DEFAULT_MAX_ATTEMPTS (5, per R3), onDeadLetter(job, err) (structured log + metric per R3), isNonRetryable(err, list) (per R5). Structured attempt log fields: job id, attempt, delay, error class, decision. Inline ASCII state diagram in the module header. All 5 workers register with the library through it, each passing its own non-retryable list. Refactor commit lands before the behavior-change commit. Required proof: shared-contract unit tests for each export (including clamp at max delay for large attempt numbers) plus one integration test per worker that the library invokes the shared policy on failure.\nHistory: none\n\nSection 2 dispositions: CQ-1 (R6) accepted; CQ-2 and CQ-3 accepted as part of R6's module contract; diagram requirement accepted as part of R6.\n\n### R7: Regression contract for the `processWebhookJob()` rewrite\nFinding: TEST-1, P1 CRITICAL, confidence 9/10, PLAN.md:17-19, reviewer: plan-eng-review (native). REGRESSION RULE.\nPlan baseline: original proposal, \"`processWebhookJob()` flow gets rewritten as part of this change. No regression test for the prior at-most-once delivery guarantee is planned\" (PLAN.md:17-19). R2 approved at-most-once preserved with partial required proof (one send on timeout/5xx; pre-send failure retries). No approved contract covers the rest of the existing behavior.\nRuntime evidence: unknown. `processWebhookJob()` and any existing tests not available in this checkout; this repo has 0 test files.\nBehavior to preserve: exactly one HTTP send per job on success, timeout and 5xx; request body, headers and signature shape; success bookkeeping (delivered mark). Intentional changes: pre-send failures retry (R2); terminal failures land in dead-letter instead of prior handling (R3).\nComparison grid:\n\n| Choice | Current | A) Characterize first, then rewrite | B) At-most-once assertions only |\n|---|---|---|---|\n| R7 regression coverage | none planned | before touching the code: characterization tests pinning request shape, headers, signature, success bookkeeping and one-send-on-timeout/5xx against the CURRENT implementation; they stay green through the rewrite; then add the R2/R3 intentional-difference assertions | after the rewrite: tests asserting exactly one send on success/timeout/5xx and retry on pre-send failure only |\n| Request shape / headers / signature | unprotected | protected | unprotected |\n| Success bookkeeping | unprotected | protected | unprotected |\n| One send on timeout/5xx (R2 proof) | approved proof | included | included |\n| Pre-send retry / dead-letter (R2, R3 proof) | approved proof | included as explicit intentional-difference tests | included |\n| R8 integration depth | pending | pending | pending |\n\nQuestion D7:\nD7 — How do we protect the existing `processWebhookJob()` behavior through the rewrite?\nProject/branch/task: `main` of the plan fixture repo, plan \"Add background job retry framework\".\nELI10: You are rewriting the code that sends webhooks to customers, and there are no tests around it. The rewrite is supposed to keep everything the same except how failures are handled. Without tests written against the current code first, there is no way to know whether the new version still sends the same request, with the same headers and signature, exactly once. The cheapest insurance is to pin the current behavior in tests before changing a line, then keep them green.\nStakes if we pick wrong: a subtly different request body or signature ships to every webhook receiver at once, or a duplicate send slips through, and the first signal is a customer complaint.\nRecommendation: A because characterization tests are cheap with AI, they are the only way to detect an unintended difference in a rewrite, and they become the permanent contract suite for the webhook path. (human: ~1 day / CC: ~20 min)\nCompleteness: A=9/10, B=6/10\nPros / cons:\nA) Characterize first, then rewrite (recommended)\n ✅ Pins request shape, headers, signature and success bookkeeping against the current code, so any unintended difference fails a test\n ✅ Intentional changes (pre-send retry, dead-letter) are written as explicit tests, so the diff between old and new behavior is documented\n ❌ Requires reading the current implementation carefully and a day of test writing before the rewrite starts\nB) At-most-once assertions only\n ✅ Covers the one guarantee the plan named as at risk\n ✅ Faster to write; no characterization pass\n ❌ Request shape, headers, signature and success bookkeeping can change silently and no test notices\nNet: pin the whole current contract for a day of work, or protect one guarantee and hope the rest survived.\nHeader: Webhook regression\nOptions:\nA) Characterize first, then rewrite (recommended)\nBefore modifying `processWebhookJob()`, write characterization tests (e.g. `processWebhookJob.test`) against the current implementation asserting: exact request body, headers and signature for a fixed payload; exactly one send on success, on timeout and on 5xx; success bookkeeping. Keep them green through the rewrite. Then add intentional-difference tests: pre-send failure schedules a retry; timeout/5xx lands in dead-letter with no second send. Flag the suite CRITICAL in the plan. R8 stays pending. Completeness 9/10. human: ~1 day / CC: ~20 min.\nB) At-most-once assertions only\nAfter the rewrite, add tests asserting exactly one send on success, timeout and 5xx, and a retry on pre-send failure. No characterization of request shape, headers, signature or success bookkeeping. R8 stays pending. Completeness 6/10. human: ~2 h / CC: ~5 min.\n\nState: approved\nActual answer: A) Characterize first, then rewrite (D7 answer, user selection)\nAccepted scope: CRITICAL regression suite. Before modifying `processWebhookJob()`, characterization tests against the current implementation assert: exact request body, headers and signature for a fixed payload; exactly one send on success, on timeout and on 5xx; success bookkeeping. They stay green through the rewrite. Then intentional-difference tests: pre-send failure schedules a retry; timeout/5xx lands in dead-letter with no second send. Sequencing: this suite is the first implementation task and gates the webhook rewrite. R8 remains pending.\nHistory: none\n\n### R8: Integration depth for library -> retryPolicy -> dead-letter\nFinding: TEST-2, P2, confidence 7/10, PLAN.md:7-9 (library hooks, per R1), reviewer: plan-eng-review (native)\nPlan baseline: no integration tests proposed. R1/R3/R6 approved \"one integration test per worker that the library invokes the shared policy on failure\" without fixing whether the library runs for real or is mocked.\nRuntime evidence: unknown. Library test harness not available.\nComparison grid:\n\n| Choice | Current | A) Real library backend in tests | B) Mocked library hooks |\n|---|---|---|---|\n| R8 integration depth | unspecified | per-worker integration tests run the actual job library against a test backend (in-process or containerized queue); assert attempt count survives a simulated worker restart, failed set contains the job with last error, strategy invoked with real attempt numbers | per-worker tests stub the library's retry hook and assert the policy is called; no restart or failed-set verification |\n| Restart persistence (R1 claim) | unverified | verified | unverified |\n| Failed-set contents (R3) | unverified | verified | asserted against a stub |\n| Approved unit tests (R4-R7) | approved | unchanged | unchanged |\n\nQuestion D8:\nD8 — Do the per-worker integration tests run the real job library, or a mocked hook?\nProject/branch/task: `main` of the plan fixture repo, plan \"Add background job retry framework\".\nELI10: The whole point of D1 was that the library remembers attempt counts across worker restarts and keeps failed jobs somewhere you can find them. A test that fakes the library cannot check either of those; it only checks that your function was called. Running the real library against a throwaway test queue is slower but proves the parts you are relying on actually behave.\nStakes if we pick wrong: the retry count resets on every deploy or the failed set is empty when you need it, and every test was green because the mock said so.\nRecommendation: A because the library's persistence and failed set are load-bearing assumptions from D1 and D3, and a mock cannot verify either; the cost is a test backend fixture the library almost certainly already ships.\nCompleteness: A=9/10, B=6/10\nPros / cons:\nA) Real library backend in tests (recommended)\n ✅ Proves attempt count survives a worker restart and that exhausted jobs are actually in the failed set with their last error\n ✅ Catches library-version behavior changes and off-by-one attempt numbering that a stub would hide\n ❌ Slower suite and a test backend fixture to maintain (in-process queue or container)\nB) Mocked library hooks\n ✅ Fast, deterministic, no external fixture\n ✅ Enough to prove the worker wiring calls the shared policy\n ❌ Restart persistence and failed-set contents stay unverified, which are exactly the guarantees D1 and D3 depend on\nNet: a slower fixture that verifies the library promises you are betting on, or fast tests that trust them.\nHeader: Integration depth\nOptions:\nA) Real library backend in tests (recommended)\nPer-worker integration tests (5) run the actual job library against a test backend (in-process or containerized queue). Assertions: strategy invoked with real attempt numbers; attempt count survives a simulated worker restart mid-backoff; after the ceiling the job is in the failed set with its last error; a listed non-retryable error is in the failed set after attempt 1. Mark [E2E]. Completeness 9/10. human: ~1 day / CC: ~30 min.\nB) Mocked library hooks\nPer-worker tests stub the library retry hook and assert the shared policy is invoked with the expected arguments. No restart or failed-set verification. Completeness 6/10. human: ~2 h / CC: ~10 min.\n\nState: approved\nActual answer: A) Real library backend in tests (D8 answer, user selection)\nAccepted scope: Five per-worker integration tests [E2E] run the actual job library against a test backend (in-process or containerized queue). Assertions: strategy invoked with real attempt numbers; attempt count survives a simulated worker restart mid-backoff; after the ceiling the job is in the failed set with its last error; a listed non-retryable error is in the failed set after attempt 1.\nHistory: none\n\nSection 3 dispositions: TEST-1 (R7) accepted, CRITICAL; TEST-2 (R8) accepted. Gaps identified: 26 (all paths, no existing coverage detectable). LLM/eval scope: none.\nTest Plan Artifact: ~/.gstack/projects/gstack-plan-count-yWJb6k/runner-main-eng-review-test-plan-20260929-175707.md\n\n### R9: Per-retry payload re-fetch and dependency-graph recompute\nFinding: PERF-1, P2, confidence 7/10, PLAN.md:22-24, reviewer: plan-eng-review (native)\nPlan baseline: original proposal, \"On every retry we re-fetch the full job payload from the database, then iterate the payload to recompute the dependency graph. Could cache the graph on the first attempt; not planned\" (PLAN.md:22-24). R1 approved: library schedules retries and carries job data. R3 approved: ceiling 5.\nRuntime evidence: unknown. Payload size, graph cost and whether the library passes job data to the handler are not verifiable here.\nComparison grid:\n\n| Choice | Current | A) Compute once, store on job | B) In-process memo | C) Leave as-is |\n|---|---|---|---|---|\n| R9 payload read per attempt | DB fetch every attempt | read payload from the library's job data if present; DB fetch only on attempt 1 otherwise | DB fetch every attempt | DB fetch every attempt |\n| R9 graph build per attempt | recompute every attempt | build on attempt 1, persist serialized graph in job data; later attempts deserialize; guard: only valid because graph is a pure function of the immutable payload, assert payload hash matches | build once per worker process, cache keyed by job id; lost on restart and not shared across workers | recompute every attempt |\n| Reads under outage (N jobs x 5 attempts) | 5N reads + 5N builds | ~N reads + N builds | 5N reads, ~N-5N builds | 5N reads + 5N builds |\n| R1/R3 approved | approved | unchanged | unchanged | unchanged |\n\nQuestion D9:\nD9 — Cache the dependency graph across retries, or recompute it on every attempt?\nProject/branch/task: `main` of the plan fixture repo, plan \"Add background job retry framework\".\nELI10: Every time a job retries, the plan reads the whole payload from the database again and rebuilds the same dependency graph from it. The payload never changes between attempts, so the answer is always the same. With up to 5 attempts, that is up to 5 reads and 5 builds per failing job, and failing jobs pile up exactly when something is already down. Building once and storing the result with the job removes almost all of that.\nStakes if we pick wrong: a downstream outage turns into a database load spike from your own retries, or you spend effort caching something that turns out to be cheap.\nRecommendation: A because the graph is derived from an immutable payload, the library already stores job data per attempt, and the guard (payload hash check) makes the cache safe; it also removes the redundant DB fetch. Confidence is medium: if payloads are tiny and the graph build is microseconds, C is acceptable and this becomes a TODO.\nCompleteness: A=9/10, B=5/10, C=4/10\nPros / cons:\nA) Compute once, store on job (recommended)\n ✅ Retries read no extra payload and build no graph; outage-time DB load drops from 5N to about N\n ✅ Survives worker restarts and works across workers because the cache lives in the job data, not in a process\n ❌ Adds serialized graph size to each job record and needs a payload-hash guard to stay correct\nB) In-process memo\n ✅ Simple to add, no change to job data shape\n ✅ Helps when the same worker process picks up the retry\n ❌ Retries usually land on a different worker or after a restart, so the memo misses most of the time and still re-fetches the payload\nC) Leave as-is\n ✅ Zero new code and no cache correctness to reason about\n ✅ Bounded at 5 attempts by D3, so the waste is finite\n ❌ Every retry storm during an outage multiplies database reads by up to 5\nNet: one persisted derived value with a hash guard, or accept a 5x read multiplier exactly when the system is least healthy.\nHeader: Graph caching\nOptions:\nA) Compute once, store on job (recommended)\nOn attempt 1, read the payload (from the library's job data if it carries it, else one DB fetch), build the dependency graph, and persist the serialized graph plus a payload hash in the job data. On later attempts, verify the hash and deserialize; on mismatch, rebuild. Required proof: unit test that attempt 2+ performs no DB fetch and no graph build when the hash matches; test that a hash mismatch triggers a rebuild. Completeness 9/10. human: ~half day / CC: ~15 min.\nB) In-process memo\nMemoize the built graph per job id in worker memory. Payload re-fetch unchanged. Required proof: test that a second attempt in the same process reuses the graph. Completeness 5/10. human: ~1 h / CC: ~5 min.\nC) Leave as-is\nRe-fetch the payload and rebuild the graph on every attempt, as the plan proposes. No new tests. Completeness 4/10. human: 0 / CC: 0.\n\nState: approved\nActual answer: A) Compute once, store on job (D9 answer, user selection)\nAccepted scope: On attempt 1, read the payload (from the library's job data if it carries it, else one DB fetch), build the dependency graph, persist the serialized graph plus a payload hash in the job data. On later attempts, verify the hash and deserialize; on mismatch, rebuild. Required proof: unit test that attempt 2+ performs no DB fetch and no graph build when the hash matches; test that a hash mismatch triggers a rebuild.\nHistory: none\n\nSection 4 dispositions: PERF-1 (R9) accepted. Suppressed: job-record growth from serialized graph for very large payloads (confidence 4).\n\nOutside Voice: CODEX_MODE disabled (codex_reviews=disabled). No outside invocation, no native replacement. outside_status: disabled. Disabled record logged via gstack-review-log (skill codex-plan-review, status skipped).\n\n### R10 — TODO proposal: webhook at-least-once upgrade (TODO-1)\n\nFinding: D2 kept `processWebhookJob()` at at-most-once, so timeouts and 5xx responses are terminal and land in the failed set on attempt 1. That was logged as an accepted shortcut (decision id f9e8dfdf-4e90-4883-9064-014d784b9405) with upgrade trigger \"receivers can dedupe on a stable delivery id\". No TODOS.md exists in the repo.\nPlan baseline: none (the plan does not mention delivery semantics beyond the rewrite).\nRuntime evidence: none available; webhook receiver code is outside this repo.\nTODO record:\n What: Move webhook delivery to at-least-once with a stable delivery id and idempotency key once receivers can dedupe.\n Why: Under at-most-once, every receiver timeout or 5xx is a lost delivery that only a manual replay recovers. At-least-once turns those into automatic retries.\n Pros: closes the biggest remaining reliability hole; reuses the retryPolicy module and failed-set tooling from this change; the characterization suite from D7 already covers the send path.\n Cons: needs receiver-side dedupe first (external dependency); duplicate deliveries during the transition if a receiver lags; signature scheme may need a delivery-id header.\n Context: after this plan lands, timeouts/5xx go to the failed set with `gstack-shortcut(dec-f9e8dfdf)` marking the cut in `processWebhookJob()`. Start there: add a `delivery_id` to the payload, publish a dedupe contract to receivers, then flip timeouts/5xx from terminal to retryable in the webhook worker's non-retryable list.\n Depends on / blocked by: receivers exposing dedupe on delivery id; this plan's T4 (webhook rewrite) merged.\n Effort: M. Priority: P3.\nComparison grid:\n A) Add to TODOS.md: keeps the upgrade trigger visible outside the code comment; zero build cost now. Completeness: kind choice.\n B) Skip: nothing recorded beyond the decision log entry and the code marker. Completeness: kind choice.\n C) Build now: extends this PR to at-least-once, contradicting D2's answer. Completeness: kind choice.\nQuestion D10 (full text):\nD10 — Record the webhook at-least-once upgrade as a TODO?\nProject/branch/task: gstack-plan-count-yWJb6k on main, retry framework plan; follow-up to D2.\nELI10: In D2 you chose to keep webhooks at \"send at most once\", so a slow or erroring receiver means that delivery is dropped into the failed set instead of retried. The fix (retry with a delivery id the receiver can dedupe) needs receiver work first. This question only decides whether we write that follow-up down in TODOS.md so it does not get lost.\nStakes if we pick wrong: skipped, the only trace is a code comment and a decision-log row; built now, this PR grows and contradicts the D2 call.\nRecommendation: A because the upgrade has an external prerequisite and a clear trigger, which is exactly what a TODO is for.\nNote: options differ in kind, not coverage — no completeness score.\nPros / cons:\nA) Add to TODOS.md (recommended)\n ✅ Creates TODOS.md with the What/Why/Context/Depends-on record above, findable by /retro and future reviews\n ✅ Zero implementation cost now; the marker in code and the TODO entry point at each other\n ❌ One more file in the repo that someone has to keep honest as work lands\nB) Skip\n ✅ No new file; the decision log and gstack-shortcut marker already carry the trigger\n ✅ Avoids a TODO nobody may pick up if receivers never add dedupe\n ❌ The trigger lives only in a comment and a JSONL row, easy to miss when receivers do change\nC) Build it now in this PR\n ✅ Ships the stronger delivery guarantee in the same change as the retry framework\n ✅ Reuses the retryPolicy module while it is fresh\n ❌ Reverses D2 and depends on receiver dedupe that does not exist yet, so duplicates would reach receivers\nNet: a TODO entry now versus relying on a code marker alone; building now is off the table until receivers can dedupe.\nHeader: Webhook TODO\nOptions:\nA) Add to TODOS.md (recommended)\nCreate TODOS.md at implementation time with the TODO record above under a `## Workers` section (P3, effort M). No product code change.\nB) Skip\nDo not create TODOS.md. The decision log entry and the gstack-shortcut marker remain the only trail.\nC) Build it now in this PR\nExtend the accepted scope to at-least-once delivery with delivery id and idempotency key; would reopen D2.\n\nState: approved\nActual answer: A) Add to TODOS.md (D10 answer, user selection)\nAccepted scope: At implementation time, create TODOS.md with a `## Workers` section holding the TODO-1 record above (What/Why/Context/Effort M/Priority P3/Depends on). No product code change beyond the file. This is a documentation task (T9 below).\nHistory: none\n\nTODOS.md updates: 1 item proposed, 1 accepted (TODO-1).\n\nApproval readiness: PASS. Checked R1 (D1=A), R2 (D2=A, accepted shortcut dec-f9e8dfdf-4e90-4883-9064-014d784b9405), R3 (D3=A), R4 (D4=A), R5 (D5=A), R6 (D6=A), R7 (D7=A, CRITICAL regression contract carried forward verbatim into T1), R8 (D8=A), R9 (D9=A), R10 (D10=A). Every accepted remedy cites its own actual user answer. No deferrals. No pending records.\n\n## Working plan (revised): Add background job retry framework\n\n### Context\nThe five background workers have no shared retry behavior today. The original plan proposed an inline exponential-backoff scheduler copied into each worker, bypassing the job library's retry hooks, with no bound on attempts, no dead-letter path, no regression coverage for the webhook worker's at-most-once guarantee, and a full payload re-fetch plus dependency-graph rebuild on every attempt. This review replaced each of those with a decision the user approved (D1 to D10). The revised plan below is the only approved version; the original is preserved above under \"Original plan (unchanged copy)\".\n\n### Architecture (D1, D3, D4, D5)\n- Use the job library's built-in retry hooks. The custom curve lives in one backoff strategy function passed to the library. First implementation step: confirm the hook accepts a delay function; if it only accepts a fixed table, keep the library's attempt bookkeeping and supply the computed delay per attempt.\n- Backoff: exponential base curve with equal jitter, delay drawn from `[curve/2, curve]`, RNG injectable for tests, configurable max-delay clamp.\n- Attempts: `DEFAULT_MAX_ATTEMPTS = 5`, per-worker override, configured alongside the strategy. Attempt N+1 is never scheduled.\n- Exhaustion or a listed non-retryable error sends the job to the library's failed/dead-letter set with the last error attached, emits one structured log line and one metric. Retention and replay are documented (T8).\n- Each of the four non-webhook workers declares its own non-retryable error list (validation, auth/permission, malformed payload; confirm the concrete classes at implementation). Listed errors dead-letter on attempt 1; everything else retries.\n\n### Webhook delivery (D2, accepted shortcut dec-f9e8dfdf-4e90-4883-9064-014d784b9405)\n- `processWebhookJob()` keeps at-most-once. Only pre-send failures retry (connection refused, DNS failure, errors raised before bytes leave). Receiver timeouts and 5xx responses are terminal and go to the failed set on attempt 1.\n- Ceiling: timeouts and 5xx are never retried. Upgrade trigger: receivers can dedupe on a stable delivery id, then move to at-least-once with an idempotency key (TODO-1).\n- Mark the cut in code at the classification point: `gstack-shortcut(dec-f9e8dfdf): timeouts/5xx never retried, upgrade when receivers can dedupe on a stable delivery id`.\n\n### Code quality (D6)\n- One `retryPolicy` module exporting `backoffStrategy(attempt, rng)`, `DEFAULT_MAX_ATTEMPTS`, `onDeadLetter(job, err)`, `isNonRetryable(err, list)`.\n- Structured attempt log fields: job id, attempt, delay, error class, decision (retry | dead-letter | success).\n- ASCII state diagram in the module header (copied below under Diagrams).\n- All five workers register through this module with their own non-retryable list. The refactor commit lands before any behavior commit.\n\n### Tests (D7, D8)\n- CRITICAL and first: a characterization suite for `processWebhookJob()` pinning exact request body, headers and signature for a fixed payload; exactly one send on success, on timeout and on 5xx; and success bookkeeping. It must be green before the rewrite starts and stay green through it. Then intentional-difference tests: pre-send failure schedules a retry; timeout/5xx land in the failed set with no second send.\n- Shared-contract unit tests per `retryPolicy` export, including the clamp and the jitter bounds with a pinned RNG.\n- Five per-worker [E2E] integration tests against the real job library on a test backend: strategy invoked with real attempt numbers; attempt count survives a simulated restart mid-backoff; exhausted job in the failed set with last error; listed non-retryable error in the failed set after attempt 1.\n- Test Plan artifact: `~/.gstack/projects/gstack-plan-count-yWJb6k/runner-main-eng-review-test-plan-20260929-175707.md` (unchanged by later decisions).\n\n### Performance (D9)\n- On attempt 1 read the payload (from the library's job data if present, else one DB fetch), build the dependency graph, persist the serialized graph plus a payload hash in the job data. Later attempts verify the hash and deserialize; a mismatch triggers a rebuild.\n\n### Follow-ups (D10)\n- Create `TODOS.md` with TODO-1 (webhook at-least-once upgrade, P3, effort M) under `## Workers`.\n\n## NOT in scope\n- At-least-once webhook delivery with idempotency keys: deferred to TODO-1 because receivers cannot dedupe yet (D2, D10).\n- A custom scheduler outside the job library: rejected in D1; the library owns attempt bookkeeping.\n- Retry budgets or circuit breakers across workers during a downstream outage: not raised by the plan; jitter plus a 5-attempt bound is the accepted mitigation (D3, D4).\n- In-process graph memoization: rejected in D9 in favor of persisting the graph on the job.\n- Per-attempt webhook signature timestamp handling: suppressed at confidence 4; revisit if the signature scheme includes a timestamp that receivers validate.\n\n## What already exists\n- The job library's retry hooks, attempt counter and failed/dead-letter set: reused, not rebuilt (D1, D3). The plan's \"same shape as the library version\" line was the tell that rebuilding added nothing.\n- The existing `processWebhookJob()` send path, headers and signature code: preserved behind the characterization suite (D7); the rewrite changes the retry envelope around it, not the request it produces.\n- The current dependency-graph builder: reused once per job on attempt 1 (D9); only the caching wrapper is new.\n- Shared-code rubric for the `retryPolicy` extraction (D6): callers = 5 workers; reuse-before-extract = no existing shared retry helper found in the plan or the (unavailable) worker files, so extraction is the reuse; helper size = four small exports; line accounting = removes five copies of the envelope, adds one module (net negative); blast radius = all five workers, mitigated by the refactor-first commit and one integration test per worker (D8).\n\n## Diagrams\n\nRetry flow through the library hooks:\n\n```\nenqueue ──▶ worker handler ──▶ success ──▶ done (log decision=success)\n │\n ▼ throws err\n isNonRetryable(err, list)? ──yes──▶ onDeadLetter(job, err) ──▶ failed set\n │ no (1 log line + 1 metric)\n ▼\n attempt < maxAttempts? ──no──▶ onDeadLetter(job, err) ──▶ failed set\n │ yes\n ▼\n delay = backoffStrategy(attempt, rng) [curve/2, curve], clamped\n │\n ▼\n library schedules attempt+1 ──▶ (restart-safe: count lives in the library)\n```\n\n`retryPolicy` state diagram (also goes in the module header):\n\n```\n ┌──────────┐ ok ┌─────────┐\n ──────▶ │ ATTEMPT n│ ────▶ │ SUCCESS │\n └──────────┘ └─────────┘\n │ err\n ▼\n ┌──────────────┐ listed ┌─────────────┐\n │ classify err │ ─────▶ │ DEAD_LETTER │ ◀──┐\n └──────────────┘ └─────────────┘ │\n │ retryable │ n == max\n ▼ │\n ┌──────────────┐ ──────────────────────────┘\n │ n < max ? │\n └──────────────┘\n │ yes\n ▼\n ┌──────────────┐ library timer ┌────────────┐\n │ BACKOFF(n) │ ──────────────▶ │ ATTEMPT n+1│\n └──────────────┘ └────────────┘\n```\n\nWebhook worker classification (at-most-once):\n\n```\nsend attempt\n ├─ pre-send failure (ECONNREFUSED, DNS, serialization) ──▶ retryable ──▶ BACKOFF\n ├─ timeout after bytes sent ──▶ terminal ──▶ DEAD_LETTER (gstack-shortcut dec-f9e8dfdf)\n ├─ 5xx ──▶ terminal ──▶ DEAD_LETTER (gstack-shortcut dec-f9e8dfdf)\n └─ 2xx ──▶ SUCCESS\n```\n\nGraph cache on job data (D9):\n\n```\nattempt 1: payload ──▶ build graph ──▶ job.data = {graph, payloadHash}\nattempt n: job.data.payloadHash == hash(payload)? ──yes──▶ deserialize graph\n └─no───▶ rebuild + overwrite\n```\n\nFiles needing inline diagrams: the `retryPolicy` module header (state diagram above); the webhook worker's classification block (the at-most-once branch table above).\n\n## Failure modes\n\n| Path | Realistic production failure | Test coverage | Error handling | User-visible? | Gap |\n|------|------------------------------|---------------|----------------|---------------|-----|\n| Library hook + strategy | Hook ignores the returned delay and uses its default table | Integration test asserts strategy invoked with real attempt numbers (T6) | Startup assertion that the hook accepted a function (T2) | Ops see wrong delays in attempt logs | covered |\n| Backoff + jitter | RNG returns out-of-range value, delay negative or above clamp | Unit tests for bounds and clamp (T2) | Clamp in `backoffStrategy` | None | covered |\n| Attempt bound | Restart mid-backoff resets the count and retries forever | Restart test (T6) | Count lives in the library, not the process | Ops see repeated attempts | covered |\n| Dead-letter | Exhausted job dropped without log or metric | Unit test log+metric fire once (T2), integration test entry in failed set (T6) | `onDeadLetter` always called on both exits | Ops alert fires | covered |\n| Non-retryable list | Validation error retried 5 times, wasting the window | Per-worker listed/unlisted tests (T5) | `isNonRetryable` short-circuit | None | covered |\n| Webhook at-most-once | Rewrite silently double-sends on timeout; receivers see duplicates | Characterization suite (T1) pins one send on timeout/5xx | Timeout/5xx classified terminal | Receivers, not our users | **critical gap in the original plan**, closed by T1 |\n| Graph cache | Payload changed after attempt 1, stale graph used | Hash-mismatch rebuild test (T7) | Hash guard | Silent if guard missing | covered |\n\nCritical gaps flagged: 1 (webhook at-most-once regression; the original plan had no test, no handling, and the failure would have been silent). Closed by T1, which gates T4.\n\n## Worktree parallelization strategy\n\nDependency table:\n\n| Step | Modules touched | Depends on |\n|------|----------------|------------|\n| S1 Webhook characterization suite (T1) | webhook worker tests | — |\n| S2 retryPolicy module + unit tests (T2) | new retryPolicy module, its tests | — |\n| S3 Migrate 4 non-webhook workers + non-retryable lists (T3, T5) | 4 worker modules, their tests | S2 |\n| S4 Webhook rewrite on retryPolicy (T4) | webhook worker | S1, S2 |\n| S5 Per-worker integration tests (T6) | integration test suite, test backend config | S3, S4 |\n| S6 Graph cache on job data (T7) | dependency-graph builder, the worker that owns it | — (S3 if that worker is one of the four) |\n| S7 Dead-letter runbook + TODOS.md (T8, T9) | docs | — |\n\nParallel lanes:\n- Lane A: S2 → S3 → S5 (shared retryPolicy module and four workers)\n- Lane B: S1 → [wait for S2] → S4 (webhook worker only)\n- Lane C: S6 (graph builder; independent unless it lives in one of the four migrated workers)\n- Lane D: S7 (docs only)\n\nExecution order: launch A, B, C, D together. B blocks at S4 until A finishes S2. Merge A and B, then run S5 against both. Merge C and D whenever green.\n\nConflict flags: the webhook worker is touched only by Lane B; the four other workers only by Lane A. If the graph builder sits inside one of the four workers, sequence Lane C after S3 instead of running it in parallel. Lane D touches no code.\n\n## Implementation Tasks\nSynthesized from this review's findings. Each task derives from a specific\nfinding above. Run with Claude Code or Codex; checkbox as you ship.\n\n- [ ] **T1 (P1, human: ~1 day / CC: ~20 min)** — webhook worker tests — Write the `processWebhookJob()` characterization suite before touching the function (CRITICAL)\n - Surfaced by: Test review — TEST-1 (R7): no regression test for the at-most-once guarantee\n - Files: webhook worker test module (paths not present in this checkout)\n - Verify: suite green on current code; asserts exact body/headers/signature, one send on success, timeout and 5xx, success bookkeeping\n- [ ] **T2 (P1, human: ~1 day / CC: ~20 min)** — retryPolicy module — Create `retryPolicy` exporting `backoffStrategy`, `DEFAULT_MAX_ATTEMPTS`, `onDeadLetter`, `isNonRetryable`, with state diagram in the header\n - Surfaced by: Architecture — ARCH-1 (R1), ARCH-2 (R3), ARCH-3 (R4); Code quality — CQ-1 (R6)\n - Files: new retryPolicy module + unit tests; confirm the library hook accepts a delay function first\n - Verify: unit tests for jitter bounds `[curve/2, curve]` with pinned RNG, clamp, log+metric fire once, `isNonRetryable` on listed/unlisted errors\n- [ ] **T3 (P1, human: ~1.5 days / CC: ~30 min)** — 4 non-webhook workers — Register each worker through `retryPolicy` and delete the copy-pasted envelopes, refactor commit before any behavior change\n - Surfaced by: Code quality — CQ-1, CQ-2, CQ-3 (R6)\n - Files: the four non-webhook worker modules\n - Verify: existing worker tests green after the refactor commit; no inline delay computation remains (grep for the old envelope)\n- [ ] **T4 (P1, human: ~1 day / CC: ~20 min)** — webhook worker — Rewrite `processWebhookJob()` on `retryPolicy` keeping at-most-once; add the `gstack-shortcut(dec-f9e8dfdf): timeouts/5xx never retried, upgrade when receivers can dedupe on a stable delivery id` marker at the classification point\n - Surfaced by: Architecture — ARCH-1 (R2, accepted shortcut); Test review — TEST-1 (R7) intentional-difference tests\n - Files: webhook worker module + its tests\n - Verify: T1 suite still green; new tests show pre-send failure schedules a retry, timeout/5xx land in failed set with no second send\n- [ ] **T5 (P1, human: ~half day / CC: ~15 min)** — 4 non-webhook workers — Declare each worker's non-retryable error list (validation, auth/permission, malformed payload; confirm classes) and pass it at registration\n - Surfaced by: Architecture — ARCH-4 (R5)\n - Files: the four non-webhook worker modules + tests\n - Verify: per worker, listed error dead-letters on attempt 1 with no retry; unlisted error retries\n- [ ] **T6 (P1, human: ~2 days / CC: ~40 min)** — integration test suite — Add five per-worker [E2E] tests against the real job library on a test backend\n - Surfaced by: Test review — TEST-2 (R8)\n - Files: integration test suite, test backend configuration\n - Verify: strategy invoked with real attempt numbers; attempt count survives simulated restart mid-backoff; exhausted job in failed set with last error; listed non-retryable in failed set after attempt 1\n- [ ] **T7 (P2, human: ~half day / CC: ~15 min)** — dependency-graph builder — Build the graph once on attempt 1 and persist it with a payload hash in job data; verify hash and deserialize on later attempts\n - Surfaced by: Performance — PERF-1 (R9)\n - Files: graph builder and the worker that owns it\n - Verify: attempt 2+ performs no DB fetch and no build when the hash matches; mismatch triggers a rebuild\n- [ ] **T8 (P2, human: ~2 h / CC: ~5 min)** — docs — Document failed-set retention and the replay procedure\n - Surfaced by: Architecture — ARCH-2 (R3): retention/replay documented\n - Files: ops/runbook doc next to the workers\n - Verify: an operator can replay one job from the failed set following the doc alone\n- [ ] **T9 (P3, human: ~15 min / CC: ~2 min)** — docs — Create `TODOS.md` with TODO-1 (webhook at-least-once upgrade) under `## Workers`\n - Surfaced by: TODOS.md updates — R10 (D10=A)\n - Files: TODOS.md\n - Verify: entry has What/Why/Context/Effort M/Priority P3/Depends on\n\nEffort assumption: tests at ~50x, module extraction at ~30x, docs at ~20x human-to-CC ratio; paths are unavailable in this checkout, so estimates assume five workers of ordinary size.\n\n## Unresolved decisions that may bite you later\nNone. D1 through D10 all answered.\n\n## Completion summary\n- Step 0: Scope Challenge — scope accepted as-is (D1 changed mechanism, not feature scope)\n- Architecture Review: 4 issues found\n- Code Quality Review: 3 issues found\n- Test Review: diagram produced, 26 gaps identified\n- Performance Review: 1 issue found\n- NOT in scope: written\n- What already exists: written\n- TODOS.md updates: 1 item proposed to user (accepted)\n- Failure modes: 1 critical gap flagged (closed by T1)\n- Unresolved decisions: 0 in this review\n- Outside voice: codex, disabled (codex_reviews disabled; recorded as outside_status disabled, no native replacement)\n- Parallelization: 4 lanes, 3 parallel / 1 sequential (Lane B waits on Lane A's S2 before the webhook rewrite)\n- Lake Score: 0/9 = 10/10 choices / answered coverage choices (D10 was a kind choice, excluded)\n- issues_found for the log: 4 + 3 + 1 + 26 = 34 (Scope Challenge SC-1 and Outside Voice reported separately)\n\n## Suppressed findings\n- Webhook signature timestamp on retried attempts: if the signature covers a timestamp, a retried pre-send failure re-signs with a new time; receivers with tight windows may reject. Confidence 4/10; the signature scheme is not visible in this checkout.\n- Job-record growth from the serialized graph for very large payloads (D9). Confidence 4/10; payload sizes unknown.\n- Retry-storm memory pressure from many simultaneous backoff timers. Confidence 4/10; the library owns timers under D1, so this is likely moot.\n\n## GSTACK REVIEW REPORT\n\n| Review | Trigger | Why | Runs | Status | Findings |\n|--------|---------|-----|------|--------|----------|\n| CEO Review | `/plan-ceo-review` | Scope & strategy | 0 | — | — |\n| Outside Review | codex via `/plan-eng-review` Outside Voice | Independent 2nd opinion | 1 | DISABLED | none (codex_reviews disabled, phase plan-review) |\n| Eng Review | `/plan-eng-review` | Architecture & tests (required) | 1 | ISSUES OPEN (this run) | 34 issues, 1 critical gaps |\n| Design Review | `/plan-design-review` | UI/UX gaps | 0 | — | — |\n| DX Review | `/plan-devex-review` | Developer experience gaps | 0 | — | — |\n\n**OUTSIDE COVERAGE:** codex, phase plan-review, disabled (codex_reviews disabled, logged 2026-09-29T18:01:01Z, source none, host claude), no findings. Native review does not substitute for outside coverage.\n\n**VERDICT:** No reviews CLEAR. Eng Review ISSUES OPEN: 34 findings mapped to 9 implementation tasks, 0 unresolved decisions, 1 critical gap closed by T1. eng review required.\n\nNO UNRESOLVED DECISIONS\n" +} diff --git a/test/fixtures/plan-create-cropped-title-batching.json b/test/fixtures/plan-create-cropped-title-batching.json new file mode 100644 index 000000000..8f970a06e --- /dev/null +++ b/test/fixtures/plan-create-cropped-title-batching.json @@ -0,0 +1,13 @@ +{ + "source": "local rerun smoke-2.1.284-1790709409 (Claude Code 2.1.284) of plan-eng-multi-finding-batching: the Create pane stayed unanswered for 1,372 s because its title row was cropped above the file row", + "cwd": "/tmp/gstack-plan-count-Z3cntL", + "screen": " ../gstack-e2e-plan-eng-batching-DINQ9m/gstack-test-plan-eng-batching.md\n\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\n 1 # Eng Review \u2014 Plan: Add background job retry framework\n 2\n 3 Review target (fixed): `/tmp/gstack-plan-count-Z3cntL/PLAN.md` on branch `main` (commit 844c6ae)\n 4 Reviewer: /plan-eng-review (Claude, session 196868-1790709430-09cebc2c), 2026-09-29\n 5 Report file: this file (user-requested destination)\n 6\n 7 ## Original plan (unchanged copy)\n 8\n 9 # Plan: Add background job retry framework\n 10\n 11 ## Architecture\n 12 We'll roll a custom exponential-backoff scheduler inline in each worker\n 13 rather than use the existing job library's built-in retry hooks. Same\n 14 shape as the library version, but we want full control over the curve.\n 15\n 16 ## Code quality\n 17 The retry envelope (compute delay, log attempt, dispatch) is duplicated\n 18 across 5 worker files with copy-pasted bodies. We will leave the\n 19 duplication for now and refactor \"later.\"\n 20\n 21 ## Tests\n 22 The existing `processWebhookJob()` flow gets rewritten as part of this\n 23 change. No regression test for the prior at-most-once delivery guarantee\n 24 is planned.\n 25\n 26 ## Performance\n 27 On every retry we re-fetch the full job payload from the database, then\n 28 iterate the payload to recompute the dependency graph. Could cache the\n 29 graph on the first attempt; not planned.\n\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\u254c\n Do you want to create gstack-test-plan-eng-batching.md?\n \u276f 1. Yes\n 2. Yes, and switch to accept edits (auto-approve file edits and common file commands) for this session; Yes, and\n always allow access to /tmp/gstack-e2e-plan-eng-batching-DINQ9m for this session (shift+tab)\n 3. No\n\n Esc to cancel \u00b7 Tab to amend\n", + "event": { + "sessionId": "16914623-f7a4-49ca-bea5-fb4b56274cb1", + "toolUseId": "toolu_01BaNDAjTWTr5bf2sdwWC4DA", + "input": { + "file_path": "/tmp/gstack-e2e-plan-eng-batching-DINQ9m/gstack-test-plan-eng-batching.md", + "content": "# Eng Review \u2014 Plan: Add background job retry framework\n\nReview target (fixed): `/tmp/gstack-plan-count-Z3cntL/PLAN.md` on branch `main` (commit 844c6ae)\nReviewer: /plan-eng-review (Claude, session 196868-1790709430-09cebc2c), 2026-09-29\nReport file: this file (user-requested destination)\n\n## Original plan (unchanged copy)\n\n# Plan: Add background job retry framework\n\n## Architecture\nWe'll roll a custom exponential-backoff scheduler inline in each worker\nrather than use the existing job library's built-in retry hooks. Same\nshape as the library version, but we want full control over the curve.\n\n## Code quality\nThe retry envelope (compute delay, log attempt, dispatch) is duplicated\nacross 5 worker files with copy-pasted bodies. We will leave the\nduplication for now and refactor \"later.\"\n\n## Tests\nThe existing `processWebhookJob()` flow gets rewritten as part of this\nchange. No regression test for the prior at-most-once delivery guarantee\nis planned.\n\n## Performance\nOn every retry we re-fetch the full job payload from the database, then\niterate the payload to recompute the dependency graph. Could cache the\ngraph on the first attempt; not planned.\n" + } + } +} \ No newline at end of file diff --git a/test/helpers/ceo-mode-option.ts b/test/helpers/ceo-mode-option.ts index 74531e4e3..321752790 100644 --- a/test/helpers/ceo-mode-option.ts +++ b/test/helpers/ceo-mode-option.ts @@ -149,7 +149,7 @@ function hasNativePostureProse(text: string, posture: RegExp): boolean { /** Finish the selected native mode packet before waiting for its answer. */ export function ceoModeSubmissionInput( visible: string, selected: NativePlanQuestionCall | undefined, targetMode: CeoMode, - transcript: PlanCountTranscript, submitted: Set, + transcript: PlanCountTranscript, submitted: Set, screenText = '', ): string | null { if (!selected || selected.answered || selected.failed || !selected.sessionId || !selected.toolUseId || transcript.status !== 'ready' || selected.questions.length < 2 || @@ -161,16 +161,28 @@ export function ceoModeSubmissionInput( const modeQuestions = selected.questions.filter(q => q.options.filter(o => modeTitle(o.label)).length >= 2); if (modeQuestions.length !== 1 || findCeoModeOption(modeQuestions[0]!.options.map((o, i) => ({index:i + 1, label:o.label})), targetMode) === null) return null; - const bar = posturePacketBar(visible); - if (!bar || !bar.answered.every(Boolean) || JSON.stringify(bar.headers) !== JSON.stringify( - selected.questions.map(q => q.header.trim().replace(/\s+/g, ' '))) || - planCountSubmissionInput(visible) !== '\r') return null; - const rawBar = [...visible.matchAll(/←[^\r\n]+✔\s*Submit\s*→/g)].at(-1)!; - const preceding = visible.slice(0, rawBar.index); - if (/```|~~~|^\s*>|\b(?:example|quoted|source)[^:\n]*:\s*$/im.test(preceding)) return null; const compact = (text: string) => text.replace(/\s+/g, ''); - const panel = compact(visible.slice(rawBar.index! + rawBar[0].length) - .replace(/^[ \t]*[│┃] ?/gm, '').replace(/^[ \t]*[●⏺] ?/gm, '')); + const quotedContext = /```|~~~|^\s*>|\b(?:example|quoted|source)[^:\n]*:\s*$/im; + const bar = posturePacketBar(visible); + let review: string; + if (bar) { + if (!bar.answered.every(Boolean) || JSON.stringify(bar.headers) !== JSON.stringify( + selected.questions.map(q => q.header.trim().replace(/\s+/g, ' '))) || + planCountSubmissionInput(visible) !== '\r') return null; + const rawBar = [...visible.matchAll(/←[^\r\n]+✔\s*Submit\s*→/g)].at(-1)!; + if (quotedContext.test(visible.slice(0, rawBar.index))) return null; + review = visible.slice(rawBar.index! + rawBar[0].length); + } else { + // A review taller than the terminal scrolls its tab bar and heading off + // the viewport (run 36606688266). The viewport must still end at the + // focused Submit prompt; the accumulated screen text then supplies the + // one complete review panel, authenticated below exactly as with a bar. + const heading = screenText.lastIndexOf('Review your answers'); + if (heading < 0 || !compact(visible).endsWith(BARLESS_SUBMIT_END) || + quotedContext.test(screenText.slice(0, heading).split('\n').slice(-3).join('\n'))) return null; + review = screenText.slice(heading); + } + const panel = compact(review.replace(/^[ \t]*[│┃] ?/gm, '').replace(/^[ \t]*[●⏺] ?/gm, '')); // Authenticate the complete review panel against native questions and // offered answers. An intended keypress or a selected-mode echo is not an ACK. let prefixes = ['Reviewyouranswers']; diff --git a/test/helpers/claude-pty-runner.ts b/test/helpers/claude-pty-runner.ts index 541373a4d..e6540e106 100644 --- a/test/helpers/claude-pty-runner.ts +++ b/test/helpers/claude-pty-runner.ts @@ -1987,7 +1987,7 @@ function conflictingDesignClosure(text: string): boolean { new RegExp(`(?:^|[.!?;]\\s+|\\n)(?:If|When|Once|Unless|Assuming|Provided)\\b[^.!?\\n]*\\b${owner}\\b`, 'i').test(text); } -function hasCompletePlanReport(expectedPlanPath: string, minimumMtime: number, maximumMtime: number, +export function hasCompletePlanReport(expectedPlanPath: string, minimumMtime: number, maximumMtime: number, allowRunHeaderForFailure = false, requiredReview?: 'Design'): boolean { if (!path.isAbsolute(expectedPlanPath)) return false; try { diff --git a/test/helpers/e2e-helpers.ts b/test/helpers/e2e-helpers.ts index 63ba664cb..4b9e1a747 100644 --- a/test/helpers/e2e-helpers.ts +++ b/test/helpers/e2e-helpers.ts @@ -229,6 +229,18 @@ export function createEvalCollector(suite: string): EvalCollector | null { } /** DRY helper to record an E2E test result into the eval collector. */ +/** Exit reasons for an API or transport failure (session-runner.ts). */ +const INFRA_EXIT_REASONS = new Set(['error_api', 'timeout_startup', 'error_output_stream']); + +/** API/transport error or CLI crash before the first model turn: INFRA, never a + * verdict on the product. Any assistant event or counted turn means the model + * ran, so its refusal, timeout or wrong answer stays an ordinary failure. */ +export function isPreTurnInfraFailure(result: Pick): boolean { + return result.costEstimate.turnsUsed === 0 + && (INFRA_EXIT_REASONS.has(result.exitReason) || /^exit_code_\d+$/.test(result.exitReason)) + && !result.transcript.some(event => event?.type === 'assistant'); +} + export function recordE2E( evalCollector: EvalCollector | null, name: string, @@ -241,9 +253,11 @@ export function recordE2E( ? `${result.toolCalls[result.toolCalls.length - 1].tool}(${JSON.stringify(result.toolCalls[result.toolCalls.length - 1].input).slice(0, 60)})` : undefined; + const passed = extra?.passed ?? (result.exitReason === 'success' && result.browseErrors.length === 0); evalCollector?.addTest({ name, suite, tier: 'e2e', - passed: result.exitReason === 'success' && result.browseErrors.length === 0, + passed, + ...(!passed && isPreTurnInfraFailure(result) ? { failure_class: 'infra' as const } : {}), duration_ms: result.duration, cost_usd: result.costEstimate.estimatedCost, transcript: result.transcript, diff --git a/test/helpers/eng-seeded-coverage.ts b/test/helpers/eng-seeded-coverage.ts index 7ccde9d7f..eac6f6c02 100644 --- a/test/helpers/eng-seeded-coverage.ts +++ b/test/helpers/eng-seeded-coverage.ts @@ -117,6 +117,9 @@ export function isEngBatchingIssueAUQ(fp: AskUserQuestionFingerprint, priorCalls return !priorCalls.some(prior => batchingIssueNumber(prior) === issue); } +// The report's target declaration field (Target / Review target / Reviewed target, optionally qualified). +const TARGET_FIELD = /^(?:Reviewed |Review )?target(?: \([^)\n]*\))?:/i; + /** A native brief can use its D number and topic while its stable R identity * lives in the required saved ledger. Count that owned choice, not a title * spelling. This does not approve the row or validate the implementation. */ @@ -160,26 +163,31 @@ function recordedBatchingIssue(call: NativePlanQuestionCall, savedPlan: string): const rawSourceNames = [...(lines[1] ?? '').matchAll(/\b[\w./-]+\.md\b/g)]; const directSource = sourceNames.length > 0 && sourceNames.every(name => name === 'PLAN.md') && new Set([...metadata.matchAll(/\bPLAN\.md:([1-9]\d*(?:[-–][1-9]\d*)?)\b/g)].map(match => match[1])).size <= 1; - const targetName = (s: string) => clean(s).replace(/^Eng(?:ineering)? review:\s*/i, '') + const targetName = (s: string) => clean(s).replace(/^Eng(?:ineering)? review\s*[:—–-]\s*/i, '') .replace(/^Plan\s*[:—–-]\s*/i, '').toLowerCase(); - const named = [...(lines[1] ?? '').matchAll(/"(Plan:\s*[^"\n]+)"|“(Plan:\s*[^”\n]+)”/g)] - .map(match => targetName(match[1] ?? match[2]!)); + const named = [...(lines[1] ?? '').matchAll(/"(Plan:\s*[^"\n]+)"|“(Plan:\s*[^”\n]+)”|\b[Pp]lan\s+"([^"\n]+)"|\b[Pp]lan\s+“([^”\n]+)”/g)] + .map(match => targetName(match[1] ?? match[2] ?? match[3] ?? match[4]!)); const titles = tokens.slice(0, start).filter(token => token.type === 'heading' && token.depth === 1); + // Target declarations are fields, whatever their list or emphasis markup. const targetFields = tokens.slice(0, start).flatMap((token, at) => { - if (token.type !== 'paragraph' || !currentHeading(at)) return []; + if ((token.type !== 'paragraph' && token.type !== 'list') || !currentHeading(at)) return []; const previous = tokens.slice(0, at).filter(t => t.type !== 'space').at(-1); const quotedContext = /\b(?:quoted|copied|historical|example|hypothetical|archived)\b[^\n]*:\s*$/i; if (previous?.type === 'paragraph' && quotedContext.test(previous.raw)) return []; - const parts = token.raw.split('\n'); - return parts.filter((line, i) => /^Reviewed target:/.test(line) && + const parts = token.raw.split('\n').map(line => line.replace(/^\s*(?:[-*+]|\d+[.)])\s+/, '').replace(/[*_]/g, '').trim()); + return parts.filter((line, i) => TARGET_FIELD.test(line) && !parts.slice(0, i).some(part => quotedContext.test(part))); }); - const namedSource = !rawSourceNames.length && named.length === 1 && titles.length === 1 && + const targetFiles = targetFields.length === 1 ? [...targetFields[0]!.matchAll(/[\w./-]*[\w-]+\.md\b/g)].map(match => match[0]) : []; + // An unsourced brief inherits the report's one current PLAN.md target; its + // ledger record still supplies the cited finding. A brief that names its plan + // must name the report title's plan, and an unfenced copy of that plan may + // add its own H1 only when it names that same plan. + const namedSource = !rawSourceNames.length && named.length <= 1 && titles.length >= 1 && titles[0]!.type === 'heading' && currentHeading(tokens.indexOf(titles[0]!)) && - /^Eng(?:ineering)? review:\s*Plan\s*[:—–-]/i.test(clean(titles[0]!.text)) && - targetName(titles[0]!.text) === named[0] && targetFields.length === 1 && - /^Reviewed target:\s*`?PLAN\.md`?(?:\s|$)/.test(targetFields[0]!) && - [...targetFields[0]!.matchAll(/\b[\w./-]+\.md\b/g)].length === 1; + targetFiles.length === 1 && targetFiles[0]!.split('/').at(-1) === 'PLAN.md' && + (named.length === 0 || /^Eng(?:ineering)? review\s*[:—–-]\s*\S/i.test(clean(titles[0]!.text)) && + titles.every(title => title.type === 'heading' && targetName(title.text) === named[0])); if (!directSource && !namedSource) return; const withdrawn = (value: string, owners: string) => new RegExp( `(?:^|[.!?;]\\s+|\\n)(?:Correction:\\s*)?(?:${owners}) (?:is|was|has been) ["“'‘]?(?:withdrawn|cancelled|canceled|rejected|superseded|resolved|closed|hypothetical|not current|no longer current)\\b`, 'i').test(prose(value, true)); @@ -265,7 +273,7 @@ function recordedBatchingIssue(call: NativePlanQuestionCall, savedPlan: string): if (questions.length !== 1) continue; const inlineBrief = fields[questions[0]!]!.slice(marker.length).trim(); const inline = Boolean(inlineBrief); - if (!inline && (namedSource || clean(fields[questions[0]! + 1] ?? '') !== clean(title))) continue; + if (!inline && clean(fields[questions[0]! + 1] ?? '') !== clean(title)) continue; const sources = [...finding[0]!.matchAll(/\b([\w./-]+\.md)(?::([1-9]\d*(?:[-–][1-9]\d*)?))?\b/g)]; if (sources.length !== 1 || sources[0]![1] !== 'PLAN.md' || !inline && !sources[0]![2] || source && sources[0]![2] !== source) continue; diff --git a/test/helpers/llm-judge.ts b/test/helpers/llm-judge.ts index da7811b47..22c24a973 100644 --- a/test/helpers/llm-judge.ts +++ b/test/helpers/llm-judge.ts @@ -512,9 +512,6 @@ export interface ArmJudgeScore { */ export const ARM_JUDGE_MODEL = CLAUDE_FRONTIER_EVAL_MODEL; -/** Bounded retry-on-malformed loop: total attempts, not extra retries. */ -export const ARM_JUDGE_ATTEMPTS = 2; - /** * Build the over-engineering rubric prompt. Exported (pure) so the free * selftest can verify prompt construction without any API call. @@ -587,10 +584,10 @@ export function parseArmJudgeResponse(raw: unknown): ArmJudgeScore { * * - Zero-diff arms are VALID scored cells: the agent built nothing, so the * score is deterministically 0/"none" — no API call. - * - Bounded retry-on-malformed: ARM_JUDGE_ATTEMPTS total attempts. callJudge - * already retries 429s internally; this loop covers malformed/refused JSON. + * - One sample, never re-asked: a malformed or refused verdict is a failed + * sample. callJudge's transport-level 429 backoff is not a verdict retry. * - `opts.call` is an injection seam so the free selftest can exercise the - * retry bound without spending API money. Defaults to the real callJudge. + * malformed path without spending API money. Defaults to the real callJudge. */ export async function armJudge( task: string, @@ -605,18 +602,10 @@ export async function armJudge( }; } const call = opts?.call ?? callJudge; - const prompt = buildArmJudgePrompt(task, diff); - let lastError: unknown; - for (let attempt = 1; attempt <= ARM_JUDGE_ATTEMPTS; attempt++) { - try { - const raw = await call>(prompt, ARM_JUDGE_MODEL); - return parseArmJudgeResponse(raw); - } catch (err) { - lastError = err; - } + const raw = await call>(buildArmJudgePrompt(task, diff), ARM_JUDGE_MODEL); + try { + return parseArmJudgeResponse(raw); + } catch (err) { + throw new Error(`armJudge: malformed verdict (never resampled) — ${err instanceof Error ? err.message : String(err)}`); } - throw new Error( - `armJudge: no well-formed verdict after ${ARM_JUDGE_ATTEMPTS} attempts — ` - + (lastError instanceof Error ? lastError.message : String(lastError)), - ); } diff --git a/test/helpers/plan-count-file-permission.ts b/test/helpers/plan-count-file-permission.ts index 8a792fa0d..824df6fa5 100644 --- a/test/helpers/plan-count-file-permission.ts +++ b/test/helpers/plan-count-file-permission.ts @@ -284,7 +284,11 @@ function currentCreatePreview(preview: string, r: any, config: string, cwd: stri if(event.name!=='Write'||`${event.sessionId}:${event.toolUseId}`!==r.pendingId||event.input?.file_path!==r.expected|| Date.parse(event.timestamp)MAX_WRITE_INPUT_BYTES) return false; - const source=event.input.content.split(/\r?\n/), rows=preview.split('\n'); + // A crop can keep the pane's file row and rule above the preview while its + // "Create file" title scrolls away. That row must name the owned path. + const header=/^ {0,3}(?![1-9]\d*(?:[ \t]|\n))(\S[^\n]*)\n[╌─━]{3,}[ \t]*\n/.exec(preview); + if(header && path.resolve(cwd,header[1]!.trim())!==r.expected) return false; + const source=event.input.content.split(/\r?\n/), rows=preview.slice(header?.[0].length ?? 0).split('\n'); const numbered:Array<{line:number;text:string}>=[]; let leading=''; for(const row of rows) { diff --git a/test/llm-judge-recommendation.test.ts b/test/llm-judge-recommendation.test.ts index 438d1da37..05af47a05 100644 --- a/test/llm-judge-recommendation.test.ts +++ b/test/llm-judge-recommendation.test.ts @@ -6,14 +6,16 @@ * negative coverage: hand-graded good/bad recommendation strings, asserted * against the same threshold the production E2E tests use (>= 4). * - * Costs ~$0.04 per run (4 Haiku calls + 3 deterministic-only fixtures). + * Each fixture is a pre-registered 3-sample judge panel: numeric substance + * gates on the panel mean, the boolean checks on a 2-of-3 majority, and an + * erroring sample fails the panel (never resampled). Costs ~$0.12 per run. * Touchfile-gated to test/helpers/llm-judge.ts so it fires on rubric * tweaks but not every test run. Runs only under EVALS=1 with an API key. */ import { expect } from 'bun:test'; import { CAPTURE_MS } from './helpers/eval-budgets'; -import { judgeRecommendation } from './helpers/llm-judge'; +import { judgePanel, judgePanelMajority, judgePanelMean, judgePanelReasoning, judgeRecommendation } from './helpers/llm-judge'; import { describeIfSelected, testIfSelected } from './helpers/e2e-helpers'; // Fixtures wrap a realistic AskUserQuestion shape so the judge sees the menu @@ -37,13 +39,24 @@ C) Hybrid — V1 client-side, V1.5 promotes to gbrain Net: optimize for V1 ship velocity vs long-term agent reusability.`; } +async function judgeRecommendationPanel(text: string) { + const samples = await judgePanel(() => judgeRecommendation(text)); + return { + present: judgePanelMajority(samples, 'present'), + commits: judgePanelMajority(samples, 'commits'), + has_because: judgePanelMajority(samples, 'has_because'), + reason_substance: judgePanelMean(samples, ['reason_substance']).reason_substance, + reasoning: judgePanelReasoning(samples), + }; +} + describeIfSelected('judgeRecommendation rubric sanity', ['llm-judge-recommendation'], () => { testIfSelected('llm-judge-recommendation', async () => { // Run all 7 fixtures sequentially in one test entry so the eval-store sees // a single result; individual assertions surface as failed expectations. // SUBSTANCE 5: option-specific reason that contrasts an alternative. - const good5 = await judgeRecommendation(buildAUQ( + const good5 = await judgeRecommendationPanel(buildAUQ( 'Recommendation: Choose C because hybrid ships V1 in gstack-only without blocking on cross-repo gbrain coordination, and locks the migration path before other agents take a hard dependency.', )); expect(good5.present).toBe(true); @@ -55,7 +68,7 @@ describeIfSelected('judgeRecommendation rubric sanity', ['llm-judge-recommendati ).toBeGreaterThanOrEqual(4); // SUBSTANCE 4: concrete option-specific reason without alternative comparison. - const good4 = await judgeRecommendation(buildAUQ( + const good4 = await judgeRecommendationPanel(buildAUQ( 'Recommendation: Choose B because client-side composition uses MCP tools that already exist in gstack and avoids any gbrain release dependency for V1.', )); expect(good4.present).toBe(true); @@ -65,7 +78,7 @@ describeIfSelected('judgeRecommendation rubric sanity', ['llm-judge-recommendati ).toBeGreaterThanOrEqual(4); // SUBSTANCE ~1: boilerplate. - const bad1 = await judgeRecommendation(buildAUQ( + const bad1 = await judgeRecommendationPanel(buildAUQ( 'Recommendation: Choose B because it is better.', )); expect(bad1.present).toBe(true); @@ -76,7 +89,7 @@ describeIfSelected('judgeRecommendation rubric sanity', ['llm-judge-recommendati ).toBeLessThan(4); // SUBSTANCE ~3: generic. - const bad3 = await judgeRecommendation(buildAUQ( + const bad3 = await judgeRecommendationPanel(buildAUQ( 'Recommendation: Choose B because it is faster.', )); expect(bad3.present).toBe(true); @@ -87,7 +100,7 @@ describeIfSelected('judgeRecommendation rubric sanity', ['llm-judge-recommendati ).toBeLessThan(4); // NO BECAUSE: missing causal connective. - const noBecause = await judgeRecommendation(buildAUQ( + const noBecause = await judgeRecommendationPanel(buildAUQ( 'Recommendation: Choose B (it has the best tradeoffs).', )); expect(noBecause.present).toBe(true); @@ -95,7 +108,7 @@ describeIfSelected('judgeRecommendation rubric sanity', ['llm-judge-recommendati expect(noBecause.reason_substance).toBe(1); // NO RECOMMENDATION: line missing entirely. - const noRec = await judgeRecommendation(`D1 — Where should the smarts live? + const noRec = await judgeRecommendationPanel(`D1 — Where should the smarts live? ELI10: ... Pros / cons: A) Server-side @@ -146,7 +159,7 @@ Net: ...`); ], ] as Array<[string, string, boolean]>; for (const [label, text, shouldPass] of crossModelCases) { - const score = await judgeRecommendation(text); + const score = await judgeRecommendationPanel(text); expect(score.present, `[cross-model:${label}] present should be true`).toBe(true); expect(score.has_because, `[cross-model:${label}] has_because should be true`).toBe(true); if (shouldPass) { @@ -175,7 +188,7 @@ Net: ...`); ['whichever fits', 'Recommendation: whichever fits the team — A or B both work.'], ]; for (const [label, text] of hedgeForms) { - const score = await judgeRecommendation(buildAUQ(text)); + const score = await judgeRecommendationPanel(buildAUQ(text)); expect(score.present, `[hedge:${label}] present should be true`).toBe(true); expect( score.commits, diff --git a/test/plan-create-combined-permission.test.ts b/test/plan-create-combined-permission.test.ts index a61523c62..639421938 100644 --- a/test/plan-create-combined-permission.test.ts +++ b/test/plan-create-combined-permission.test.ts @@ -1,4 +1,4 @@ -import {expect, test} from 'bun:test'; +import {describe, expect, test} from 'bun:test'; import * as fs from 'node:fs'; import * as os from 'node:os'; import * as path from 'node:path'; @@ -6,6 +6,7 @@ import {createFilePermissionRecorder, recordFilePermission, currentFilePermissio import {readPlanCountTranscript} from './helpers/plan-count-transcript'; import {createPlanCountPermissionGuard} from './helpers/claude-pty-runner'; import captures from './fixtures/plan-create-combined-permission-70b.json'; +import croppedTitle from './fixtures/plan-create-cropped-title-batching.json'; function fixture(captured: typeof captures[number]) { const dir=fs.mkdtempSync(path.join(os.tmpdir(),'create-combined-')); @@ -101,3 +102,50 @@ for(const captured of captures) { }finally{f.close();} }); } + +describe('Create pane cropped below its title (plan-eng-multi-finding-batching, CLI 2.1.284)', () => { + function cropped() { + const dir=fs.mkdtempSync(path.join(os.tmpdir(),'create-cropped-')); + const cwd=path.join(dir,path.basename(croppedTitle.cwd)), config=path.join(dir,'config');fs.mkdirSync(cwd); + const originalDir=path.dirname(croppedTitle.event.input.file_path); + const expected=path.join(dir,path.basename(originalDir),path.basename(croppedTitle.event.input.file_path)); + fs.mkdirSync(path.dirname(expected)); + const {sessionId,toolUseId:id}=croppedTitle.event, timestamp=new Date().toISOString(); + const journal=path.join(config,'projects','owned',sessionId+'.jsonl');fs.mkdirSync(path.dirname(journal),{recursive:true}); + const recorder=createFilePermissionRecorder(cwd,config,expected)!; + const input={...croppedTitle.event.input,file_path:expected}; + fs.writeFileSync(journal,JSON.stringify({cwd,sessionId,isSidechain:false,timestamp, + message:{role:'assistant',content:[{type:'text',text:'Writing the review report.'},{type:'tool_use',id,name:'Write',input}]}})+'\n'); + recordFilePermission(JSON.stringify({hook_event_name:'PreToolUse',tool_name:'Write',session_id:sessionId, + tool_use_id:id,cwd,transcript_path:journal,tool_input:input}),recorder.file,cwd,config,expected); + // The relative file row keeps its captured sibling layout; only the footer's + // absolute directory is rebound to this fixture. + const screen=croppedTitle.screen.replaceAll(originalDir,path.dirname(expected)); + const read=(s=screen)=>currentFilePermissionEpoch(recorder.file,expected,cwd,config,Date.now()-1000,readPlanCountTranscript(config,cwd),s); + return {screen,read,id:`${sessionId}:${id}`,close(){recorder.dispose();fs.rmSync(dir,{recursive:true,force:true});}}; + } + + test('the captured pane shows its file row and rule but not the Create file title', () => { + expect(croppedTitle.screen).not.toMatch(/(?:^|\n) {0,3}Create file[ \t]*\n/); + expect(croppedTitle.screen.split('\n')[0]).toBe(' ../gstack-e2e-plan-eng-batching-DINQ9m/gstack-test-plan-eng-batching.md'); + }); + + test('the owned file row binds the pending Write and grants it once', () => { + const f=cropped();try { + const epoch=f.read();expect(epoch?.pendingId).toBe(f.id); + const guard=createPlanCountPermissionGuard(); + expect(guard(f.screen,'',epoch)).toBe('grant'); + expect(guard(f.screen,'',epoch)).toBe('handled'); + }finally{f.close();} + }); + + for(const [name,change] of [ + ['a file row naming another file',(s:string)=>s.replace('/gstack-test-plan-eng-batching.md\n','/other.md\n')], + ['a file row in another directory',(s:string)=>s.replace(' ../gstack-e2e-plan-eng-batching-DINQ9m/',' ../elsewhere/')], + ['an edited preview row',(s:string)=>s.replace('We will leave the','We will fix the')], + ] as const) test(`the cropped pane rejects ${name}`,()=>{ + const f=cropped();try { + const screen=change(f.screen);expect(screen).not.toBe(f.screen);expect(f.read(screen)).toBeFalsy(); + }finally{f.close();} + }); +}); diff --git a/test/plan-review-report-recording.test.ts b/test/plan-review-report-recording.test.ts index d6e6c5f8f..28354d1d6 100644 --- a/test/plan-review-report-recording.test.ts +++ b/test/plan-review-report-recording.test.ts @@ -4,7 +4,7 @@ import * as fs from 'node:fs'; import * as os from 'node:os'; import * as path from 'node:path'; import { CAPTURE_LONG_MS } from './helpers/eval-budgets'; -import { recordE2E } from './helpers/e2e-helpers'; +import { isPreTurnInfraFailure, recordE2E } from './helpers/e2e-helpers'; import { EvalCollector, isFinalizedEvalResultFile, listEvalJsonFiles, type EvalTestEntry } from './helpers/eval-store'; import { OFFICE_HOURS_BUN_GRACE_MS, runRecordedOfficeHoursAttempt } from './helpers/office-hours-attempt'; import { isPaidTestFile } from './helpers/paid-test-set'; @@ -257,3 +257,37 @@ test('report deadline aborts, records once, cleans up and ignores late completio test('report recording controls stay outside the paid test filename patterns', () => { expect(isPaidTestFile('test/plan-review-report-recording.test.ts')).toBe(false); }); + +// A pre-turn API/transport failure is INFRA; once the model has run, a failure +// keeps its ordinary class. +function runnerResult(exitReason: string, turnsUsed = 0, transcript: any[] = [{ type: 'system', subtype: 'init' }]): any { + return { exitReason, transcript, toolCalls: [], browseErrors: [], duration: 1, output: '', + costEstimate: { inputChars: 1, outputChars: 0, estimatedTokens: 0, estimatedCost: 0, turnsUsed } }; +} +function recordedClass(result: any, extra?: Partial) { + const collector = new EvalCollector('e2e'); + recordE2E(collector, 'infra-probe', 'Infra probe', result, extra); + return (collector as any).tests.at(-1)?.failure_class; +} + +test.each(['error_api', 'timeout_startup', 'error_output_stream', 'exit_code_1'])( + 'a %s before the first model turn is recorded as infra', exitReason => { + expect(isPreTurnInfraFailure(runnerResult(exitReason))).toBe(true); + expect(recordedClass(runnerResult(exitReason))).toBe('infra'); + }); + +test.each([ + ['a timeout after model work', runnerResult('timeout')], + ['max turns', runnerResult('error_max_turns', 24)], + ['an API error after a turn', runnerResult('error_api', 3)], + ['an API error after an assistant message', runnerResult('error_api', 0, [{ type: 'assistant', message: { content: [{ type: 'text', text: 'I cannot help with that.' }] } }])], + ['a successful run', runnerResult('success', 2)], +] as const)('%s is not infra', (_name, result) => { + expect(isPreTurnInfraFailure(result)).toBe(false); + expect(recordedClass(result)).toBeUndefined(); +}); + +test('an explicit pass or class from the caller wins', () => { + expect(recordedClass(runnerResult('error_api'), { passed: true })).toBeUndefined(); + expect(recordedClass(runnerResult('error_api'), { failure_class: 'assertion' })).toBe('assertion'); +}); diff --git a/test/skill-e2e-plan-ceo-mode-routing.test.ts b/test/skill-e2e-plan-ceo-mode-routing.test.ts index 3b099f88d..c1b128db5 100644 --- a/test/skill-e2e-plan-ceo-mode-routing.test.ts +++ b/test/skill-e2e-plan-ceo-mode-routing.test.ts @@ -260,7 +260,7 @@ describeE2E('/plan-ceo-review mode routing (gate)', () => { } const currentInput = await session.currentScreen(); capture('awaiting_posture', currentInput, transcript); - const modeSubmit = ceoModeSubmissionInput(currentInput, question.nativeCall, c.mode, transcript, submittedModePackets); + const modeSubmit = ceoModeSubmissionInput(currentInput, question.nativeCall, c.mode, transcript, submittedModePackets, session.visibleText()); if (modeSubmit !== null) { session.send(modeSubmit); continue; } const pendingQuestion = readPendingQuestion(session.pendingQuestionFile, fixture.cwd, session.hermeticConfigDir, selectionStartedAt, transcript); diff --git a/test/skill-e2e-plan-eng-multi-finding-batching.test.ts b/test/skill-e2e-plan-eng-multi-finding-batching.test.ts index 7c94803f1..198641581 100644 --- a/test/skill-e2e-plan-eng-multi-finding-batching.test.ts +++ b/test/skill-e2e-plan-eng-multi-finding-batching.test.ts @@ -35,6 +35,7 @@ import { engStep0Boundary, engSetupAUQ, engFirstReviewAUQ, + hasCompletePlanReport, } from './helpers/claude-pty-runner'; import { FORCING_BATCHING_ENG } from './fixtures/forcing-finding-seeds'; import { createEngBatchingIssueCounter } from './helpers/eng-seeded-coverage'; @@ -53,6 +54,7 @@ describeE2E('/plan-eng-review multi-finding batching regression (periodic)', () test( `4-finding plan emits >= ${FLOOR} review-phase AskUserQuestions (no batching)`, async () => { + const startedAt = Date.now(); const tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'gstack-e2e-plan-eng-batching-')); const planPath = path.join(tmpDir, 'gstack-test-plan-eng-batching.md'); const followUpPrompt = FORCING_BATCHING_ENG.replaceAll(FIXTURE_PLAN_PATH, planPath); @@ -85,8 +87,12 @@ describeE2E('/plan-eng-review multi-finding batching regression (periodic)', () // review decisions exist, a batching regression can no longer occur // in this attempt; stop instead of letting the review run to the // ceiling (run 36385945043: floor at 6m41s, ceiling at 12m13s). + // A completed report ends the review, so the count is final there: + // grade it now rather than waiting out the session (run 36606688266 + // wrote its report at 1,248 s and closed at 1,318 s). isCollectionComplete: (_transcript, fingerprints) => - fingerprints.filter(fp => !fp.preReview && !fp.administrative).length >= FLOOR, + fingerprints.filter(fp => !fp.preReview && !fp.administrative).length >= FLOOR || + hasCompletePlanReport(planPath, startedAt, Date.now()), reviewCountCeiling: N + 3, // hard cap above floor + tolerance // Supplied prerequisites: routing setup and cross-project learnings are // already declined, so the attempt starts at the review (setup answers diff --git a/test/skill-e2e-review.test.ts b/test/skill-e2e-review.test.ts index 872382737..b8e452572 100644 --- a/test/skill-e2e-review.test.ts +++ b/test/skill-e2e-review.test.ts @@ -8,6 +8,7 @@ import { createEvalCollector, finalizeEvalCollector, } from './helpers/e2e-helpers'; import { extractSkillSections, REVIEW_E2E_SECTIONS } from './helpers/skill-fixture'; +import { expectContract } from './helpers/eval-store'; import { spawnSync } from 'child_process'; import * as fs from 'fs'; import * as path from 'path'; @@ -291,7 +292,8 @@ Important: The design checklist should catch issues like blacklisted fonts, smal console.log(`Design review detected ${detected}/7 planted checklist signals; detector rows surfaced: ${detectorSeen}`); expect(detected).toBeGreaterThanOrEqual(4); // the LLM-checklist bar, unchanged by the detector - expect(detectorSeen).toBe(true); // the fake engine's rows are deterministic; the review must carry them + // The fake engine's rows are deterministic; carrying them is the contract. + expectContract(detectorSeen, 'review-design-lite: the review omitted the mechanical detector rows', { collector: evalCollector, name: '/review design lite' }); } }, CAPTURE_MS + REVIEW_FINALIZE_MS); }); diff --git a/test/skill-e2e-shared-libs-periodic.test.ts b/test/skill-e2e-shared-libs-periodic.test.ts index bc32551c8..78a764bdb 100644 --- a/test/skill-e2e-shared-libs-periodic.test.ts +++ b/test/skill-e2e-shared-libs-periodic.test.ts @@ -4,7 +4,7 @@ import * as fs from 'node:fs'; import * as path from 'node:path'; import { CAPTURE_LONG_MS } from './helpers/eval-budgets'; import { describeE2ETier, e2eTierEnabled } from './helpers/e2e-gate'; -import { EvalCollector } from './helpers/eval-store'; +import { EvalCollector, expectContract } from './helpers/eval-store'; import { sharedLibsPlanExcerpt } from './helpers/shared-libs-plan-excerpt'; import { createSharedPlanReuseSelector } from './helpers/shared-libs-plan-actor'; import { @@ -51,17 +51,24 @@ async function assertJudgment(report: string, criteria: Record) for (const key of Object.keys(criteria)) expect(judgment.checks[key], `${key}: ${judgment.reasoning}`).toBe(true); } -function assertReadOnly(f: SharedLibsFixture, before: Record, result: any) { +function assertReadOnly(f: SharedLibsFixture, before: Record, result: any, name: string) { + // Read-only is the contract of every audit, even where the recommendation + // itself is a tolerated judgment call. + const record = { collector, name }; result.providerRequests = readRequests(f); - expect(sharedReadOnlyViolations(result.toolCalls, result.providerRequests)).toEqual([]); + const violations = sharedReadOnlyViolations(result.toolCalls, result.providerRequests); + expectContract(violations.length === 0, `read-only: disallowed commands or provider requests ${JSON.stringify(violations)}`, record); const expected = { ...before }, after = snapshotFixture(f.root); // Only the source-provider instrumentation can change. Snapshot the outer // fixture as well as the repository; inspect commands for writes beyond it. delete expected[path.relative(f.root, f.trace)]; delete after[path.relative(f.root, f.trace)]; - expect(after).toEqual(expected); - expect(fs.existsSync(f.hookTrace) ? fs.readFileSync(f.hookTrace, 'utf8') : '').toBe(''); - expect(fs.readdirSync(f.state)).toEqual([]); + const changed = [...new Set([...Object.keys(expected), ...Object.keys(after)])].filter(file => expected[file] !== after[file]); + expectContract(changed.length === 0, `read-only: fixture files changed: ${changed.join(', ')}`, record); + const hookTrace = fs.existsSync(f.hookTrace) ? fs.readFileSync(f.hookTrace, 'utf8') : ''; + expectContract(hookTrace === '', `read-only: a configured hook ran: ${hookTrace.slice(0, 500)}`, record); + const state = fs.readdirSync(f.state); + expectContract(state.length === 0, `read-only: gstack state written: ${state.join(', ')}`, record); } describeE2E('Shared-code opportunity and coordination judgment (periodic)', () => { @@ -77,7 +84,7 @@ describeE2E('Shared-code opportunity and coordination judgment (periodic)', () = const before = snapshotFixture(f.root); await judgedCapture(attempt, 'empty', 'shared-libs-opportunity-judgment', () => runSharedCapture(f, 'shared-libs-opportunity-judgment', `Run /deslop-shared-libs using ${instructions} and return the report.`, attempt), async result => { - assertReadOnly(f, before, result); + assertReadOnly(f, before, result, 'shared-libs-opportunity-judgment'); await assertJudgment(result.output, { valid_empty: 'There is only README, .gitignore and a unique one-line src/version.ts, and successful empty PR results. It reports no worthwhile sharing opportunities and does not fabricate callers or blame unavailable history/API access.', }); @@ -110,7 +117,7 @@ describeE2E('Shared-code opportunity and coordination judgment (periodic)', () = const before = snapshotFixture(f.root); await judgedCapture(attempt, 'opportunity', 'shared-libs-opportunity-judgment', () => runSharedCapture(f, 'shared-libs-opportunity-judgment', `Run /deslop-shared-libs using ${instructions}. Review the active TypeScript and Python areas and return the requested report.`, attempt), async result => { - assertReadOnly(f, before, result); + assertReadOnly(f, before, result, 'shared-libs-opportunity-judgment'); expect(result.output).toContain(f.tip.slice(0, 7)); expect(result.output).toContain('lib/retry-after.ts'); expect(JSON.stringify(result.transcript ?? result.toolCalls)).toMatch(/inventory\.py|search\.py|negative inventory|src\/\*\.py/); @@ -143,7 +150,7 @@ describeE2E('Shared-code opportunity and coordination judgment (periodic)', () = const before = snapshotFixture(f.root); await judgedCapture(attempt, 'audit', 'shared-libs-pr-coverage', () => runSharedCapture(f, 'shared-libs-pr-coverage', `Run /deslop-shared-libs using ${instructions}. Recent PR 7 mentions https://github.com/fixture/shared-libs/pull/42 as related work. Return the report after checking coordination within the skill's budget.`, attempt), async result => { - assertReadOnly(f, before, result); + assertReadOnly(f, before, result, 'shared-libs-pr-coverage'); const requests = readRequests(f).filter(row => row.tool === 'gh' || row.tool === 'curl'); const endpoints = requests.map(row => row.endpoint || ''); expect(endpoints.some(endpoint => /\/pulls\/42\/files/.test(endpoint))).toBe(true);