diff --git a/test/ceo-email-ledger-36385945043.test.ts b/test/ceo-email-ledger-36385945043.test.ts new file mode 100644 index 000000000..c304e0c59 --- /dev/null +++ b/test/ceo-email-ledger-36385945043.test.ts @@ -0,0 +1,97 @@ +/** Free replay of run 36385945043's two CEO email-seed throws (FAN-1, ERR-1); no model calls. */ +import { expect, test } from 'bun:test'; +import * as fs from 'node:fs'; +import * as path from 'node:path'; +import captured from './fixtures/ceo-email-ledger-36385945043.json'; +import { ceoPaymentFinding, createCeoPaymentFindingCounter } from './helpers/ceo-payment-findings'; +import { ceoFirstReviewAUQ, nativePlanCallFingerprint } from './helpers/claude-pty-runner'; +import type { NativePlanQuestionCall } from './helpers/plan-count-transcript'; + +// The seed is regenerated from the paid test's unchanged source for each captured plan path. +const source = fs.readFileSync(path.join(import.meta.dir, 'skill-e2e-plan-ceo-finding-count.test.ts'), 'utf8'); +const seedSource = source.slice(source.indexOf('const planCeo5Findings'), source.indexOf('\nconst planCeo2PairedFindings')); +const seedOf = new Function(new Bun.Transpiler({ loader: 'ts' }).transformSync(seedSource) + '\nreturn planCeo5Findings;')() as (plan: string) => string; +const [fan, err] = captured.attempts; +const LEDGER = '## Decision ledger\n\n| ID and owner | Contract and evidence | Current | Proposed | Status | Exact approval and scope |\n|---|---|---|---|---|---|\n'; +// RECONSTRUCTED saved plans: ledger rows (and FAN-1's saved grid) from the rendered Edit/Write diffs. +const ledgerPlan = (row: string) => `# Payment Processing Integration (reconstructed)\n\n${LEDGER}${row}\n`; +const fp = (call: unknown) => nativePlanCallFingerprint(structuredClone(call) as NativePlanQuestionCall, 0, false); +const withQuestion = (call: NativePlanQuestionCall, change: (q: NativePlanQuestionCall['questions'][number]) => void) => { + const copy = structuredClone(call), q = copy.questions[0]!, answer = copy.answers![q.question]!; + change(q); copy.answers = { [q.question]: answer }; + return copy; +}; +function fanPayloadPlan(row = fan!.reconstructedLedgerRow!) { + const q = fan!.call.questions[0]!; + return ledgerPlan(row) + ['', '## currentDecision (FAN-1)', 'Commitment comparison:', '', '```text', ...fan!.reconstructedGrid!, '```', '', + `Question: ${q.question}`, '', `Header: ${q.header}`, + ...q.options.flatMap((o, i) => [`${'ABC'[i]}) ${o.label}`, o.description!]), ''].join('\n'); +} + +test('fixture provenance names the run, shards, attempts and reconstructed parts', () => { + expect(captured.provenance.run).toContain('36385945043'); + expect(captured.provenance.qualification).toContain('RECONSTRUCTED'); + for (const attempt of captured.attempts) { + expect(attempt.observedError).toContain('Unsupported current CEO decision'); + expect(attempt.observedError).toEndWith(`${attempt.call.sessionId}:${attempt.call.toolUseId}`); + expect(attempt.attemptDir).toContain('skill-e2e-plan-ceo-finding-count/pty-count/'); + } +}); + +test('replay: the email explanation-defect predicate is the first seed-classifier predicate to fail for both questions', () => { + const traceFan: string[] = [], traceErr: string[] = []; + expect(ceoPaymentFinding(fp(fan!.call), seedOf(fan!.seedPlanPath), ledgerPlan(fan!.reconstructedLedgerRow!), traceFan)).toBeNull(); + expect(traceFan).toContain('email@FAN-1: seedPlan=yes rowSubject=yes rowDefect=yes questionSubject=yes explanationDefect=no operativeOption=yes proposal=yes'); + expect(ceoPaymentFinding(fp(err!.call), seedOf(err!.seedPlanPath), ledgerPlan(err!.reconstructedLedgerRow!), traceErr)).toBeNull(); + // ERR-1's rendered row proposes nothing itself and no currentDecision (ERR-1) payload was rendered. + expect(traceErr).toContain('email@ERR-1: seedPlan=yes rowSubject=yes rowDefect=yes questionSubject=yes explanationDefect=no operativeOption=yes proposal=no'); +}); + +test('the fail-closed throw names the header, the question start and the matched obligation predicates', () => { + const counter = createCeoPaymentFindingCounter(seedOf(fan!.seedPlanPath), () => ledgerPlan(fan!.reconstructedLedgerRow!), ceoFirstReviewAUQ); + let message = ''; + try { counter.isReviewAUQ(fp(fan!.call)); } catch (error) { message = String(error); } + expect(message).toContain(`Unsupported current CEO decision; cannot exclude it from the 4–7 count: ${fan!.call.sessionId}:${fan!.call.toolUseId}`); + expect(message).toContain('\nheader: FAN-1 mail leg\n'); + expect(message).toContain(`\nquestion: ${fan!.call.questions[0]!.question.slice(0, 200)}\n`); + expect(message).toContain('obligation predicates: '); + expect(message).toContain('email@FAN-1: seedPlan=yes rowSubject=yes rowDefect=yes questionSubject=yes explanationDefect=no'); +}); + +test('RECONSTRUCTED: with its saved currentDecision payload FAN-1 counts as a recorded decision, not a seed finding', () => { + // The rendered diffs show this payload saved before the question. Replaying it + // does not reproduce the throw, so the real saved plan must have differed. + const plan = fanPayloadPlan(); + const counter = createCeoPaymentFindingCounter(seedOf(fan!.seedPlanPath), () => plan, ceoFirstReviewAUQ); + expect(counter.isReviewAUQ(fp(fan!.call))).toBe(true); + expect(counter.trace).toMatchObject([{ kind: 'recorded-decision', ledgerId: 'FAN-1', phase: 'currentDecision (FAN-1)' }]); +}); + +test('negative control: an unrelated question earns no floor credit', () => { + const unrelated = withQuestion(fan!.call as NativePlanQuestionCall, q => { + q.question = q.question.replace('D2 — FAN-1: What should the handler do when the inline receipt send raises after the user update?', + 'D2 — UX-1: Which brand colour should the receipt email use?'); + q.header = 'UX-1 colour'; + }); + expect(ceoPaymentFinding(fp(unrelated), seedOf(fan!.seedPlanPath), fanPayloadPlan())).toBeNull(); + const counter = createCeoPaymentFindingCounter(seedOf(fan!.seedPlanPath), () => fanPayloadPlan(), ceoFirstReviewAUQ); + expect(() => counter.isReviewAUQ(fp(unrelated))).toThrow(/Unsupported current CEO decision[\s\S]*header: UX-1 colour/); +}); + +test('negative control: an email question whose ledger row says the failure is already rescued is not the seed finding', () => { + const rescued = fan!.reconstructedLedgerRow!.replace('| Mail exception escapes handler; webhook 500; Stripe retries committed payment |', + '| Mail failures are already rescued in the handler after commit; webhook returns 200 |'); + expect(rescued).not.toBe(fan!.reconstructedLedgerRow); + const trace: string[] = []; + expect(ceoPaymentFinding(fp(fan!.call), seedOf(fan!.seedPlanPath), ledgerPlan(rescued), trace)).toBeNull(); + expect(trace.find(line => line.startsWith('email@FAN-1'))).toContain('rowDefect=no'); +}); + +test('negative control: a ledger ID whose row belongs to another seed is not the email finding', () => { + const dispatcherRow = '| FAN-1 (owner: Section 1) | PLAN.md 106-108: new handler bypasses the dispatcher | New class bypasses `WebhookDispatcher` | A) register through WebhookDispatcher; B) keep bypass | unresolved | pending |'; + const trace: string[] = []; + expect(ceoPaymentFinding(fp(fan!.call), seedOf(fan!.seedPlanPath), ledgerPlan(dispatcherRow), trace)).toBeNull(); + expect(trace.some(line => line.startsWith('email@FAN-1') && / rowSubject=no /.test(line) || line.startsWith('dispatcher@FAN-1') && / questionSubject=no /.test(line))).toBe(true); + const counter = createCeoPaymentFindingCounter(seedOf(fan!.seedPlanPath), () => ledgerPlan(dispatcherRow), ceoFirstReviewAUQ); + expect(() => counter.isReviewAUQ(fp(fan!.call))).toThrow(/Unsupported current CEO decision/); +}); diff --git a/test/fixtures/ceo-email-ledger-36385945043.json b/test/fixtures/ceo-email-ledger-36385945043.json new file mode 100644 index 000000000..de1e73575 --- /dev/null +++ b/test/fixtures/ceo-email-ledger-36385945043.json @@ -0,0 +1,96 @@ +{ + "provenance": { + "run": "garrytan/gstack actions run 36385945043 (evals-periodic, 2026-09-28), slice 3, test/skill-e2e-plan-ceo-finding-count.test.ts, the two attempts that threw", + "qualification": "Each failing native call and its observed error are retained actual bytes from observation.json. The run did not upload the saved plan: ledger rows and the FAN-1 commitment grid are RECONSTRUCTED from the Edit/Write diffs rendered in terminal.visible.log (FAN-1 row added before the question rendered; ERR-1 row as last rendered; no currentDecision (ERR-1) payload appears in the rendered diffs). Seed plans are regenerated from the unchanged test source. Neither attempt receives verdict credit." + }, + "attempts": [ + { + "attemptDir": "shards/skill-e2e-plan-ceo-finding-count/pty-count/local-e9a41bfc-4e61-4779-90bf-ba4b3d2a5907/plan-ceo-review-1790577236907-wZBZ6W", + "observedError": "Unsupported current CEO decision; cannot exclude it from the 4–7 count: 1b8d3018-aaaf-48ff-a642-4b3d732cb071:toolu_01GR9ntLW4U3vpME6hDzARpz", + "seedPlanPath": "/home/runner/.cache/gstack-paid-shard-DAsxIf/tmp/gstack-e2e-plan-ceo-s9FbGu/gstack-test-plan-ceo.md", + "call": { + "sessionId": "1b8d3018-aaaf-48ff-a642-4b3d732cb071", + "toolUseId": "toolu_01GR9ntLW4U3vpME6hDzARpz", + "questions": [ + { + "question": "D2 — FAN-1: What should the handler do when the inline receipt send raises after the user update?\nProject/branch/task: main; CEO review of the Payment Processing Integration plan (HOLD SCOPE).\nELI10: The handler marks the user paid, then emails a receipt. The email client already does the right thing on failure: it writes a durable \"retry this notification\" record and raises. The plan then lets that exception fly out of the handler, so the web layer answers 500 and Stripe re-sends the whole event for up to three days. Every re-send re-runs a payment that already committed and appends another failed-notification record, and the on-call alert reads \"webhook failed\" for what is really \"email failed\". If the email call happens to sit inside the database transaction, the failure also un-pays the user.\nStakes if we pick wrong: a paid customer who looks unpaid, a retry storm against your mail provider during its outage, and an alert that sends on-call down the wrong runbook.\nRecommendation: A because it maps to \"zero silent failures\" and \"name each error's class\": commit first, rescue only the mail client's named exceptions, log a structured warning with event/user/PaymentIntent ids, return success so dedup records completion, and leave notification retry to the runbook that already owns it.\nCompleteness: A=10/10, B=3/10, C=6/10\nNet: A keeps one retry path per concern (Stripe for payment, runbook for receipts); B leans on Stripe as an accidental mail retry loop and risks rolling back the payment; C fixes the common timeout case and leaves provider 5xx on the wrong path.", + "header": "FAN-1 mail leg", + "options": [ + { + "label": "Commit, then rescue named mail errors, return 200 (recommended)", + "description": "Effort S, risk low, reuse high (mail client's durable attempt record + runbook), verification: unit tests for MailTimeout and a delivery error asserting user stays paid, handler returns success, warning logged with event/user/PI ids, and no exception escapes; plus one test that a DB error still propagates. ✅ Payment commit and receipt send become independent: a mail outage can never un-pay a user or trigger Stripe retries. ✅ On-call gets the existing mail failure-rate alert and the notification backlog, exactly what the runbook expects, with no false \"webhook failed\" page. ❌ Adds a small rescue block the implementer must keep in sync with the mail client's exception classes; a new exception type there would escape until named." + }, + { + "label": "Keep as written: no error handling on the email leg", + "description": "Effort S (zero implementation work), risk high, reuse low, verification: none planned. ✅ Nothing to write; the plan text stands as is. ✅ Stripe's retries mean a transient mail blip eventually gets a receipt out without the runbook. ❌ Every mail failure becomes a webhook 500, a paged alert for the wrong cause, and repeated handler runs for an already-committed payment; if the send is inside the transaction, the payment rolls back." + }, + { + "label": "Rescue MailTimeout only; other mail errors still propagate", + "description": "Effort S, risk medium, reuse high, verification: test for MailTimeout rescued and for a delivery error propagating. ✅ Covers the most frequent failure (the 1s deadline) with the smallest rescue. ✅ Keeps unfamiliar provider errors loud during rollout, which some teams prefer while learning the new path. ❌ Provider 5xx and network errors still take the wrong retry path, so the retry-storm and false-alert problems remain for exactly the outages that matter most." + } + ], + "multiSelect": false + } + ], + "answered": true, + "failed": false, + "answers": { + "D2 — FAN-1: What should the handler do when the inline receipt send raises after the user update?\nProject/branch/task: main; CEO review of the Payment Processing Integration plan (HOLD SCOPE).\nELI10: The handler marks the user paid, then emails a receipt. The email client already does the right thing on failure: it writes a durable \"retry this notification\" record and raises. The plan then lets that exception fly out of the handler, so the web layer answers 500 and Stripe re-sends the whole event for up to three days. Every re-send re-runs a payment that already committed and appends another failed-notification record, and the on-call alert reads \"webhook failed\" for what is really \"email failed\". If the email call happens to sit inside the database transaction, the failure also un-pays the user.\nStakes if we pick wrong: a paid customer who looks unpaid, a retry storm against your mail provider during its outage, and an alert that sends on-call down the wrong runbook.\nRecommendation: A because it maps to \"zero silent failures\" and \"name each error's class\": commit first, rescue only the mail client's named exceptions, log a structured warning with event/user/PaymentIntent ids, return success so dedup records completion, and leave notification retry to the runbook that already owns it.\nCompleteness: A=10/10, B=3/10, C=6/10\nNet: A keeps one retry path per concern (Stripe for payment, runbook for receipts); B leans on Stripe as an accidental mail retry loop and risks rolling back the payment; C fixes the common timeout case and leaves provider 5xx on the wrong path.": "Commit, then rescue named mail errors, return 200 (recommended)" + }, + "unansweredQuestionIndices": [], + "answeredAt": "2026-09-28T06:40:44.000Z" + }, + "reconstructedLedgerRow": "| FAN-1 (owner: Section 2, handler implementer) | PLAN.md 114-116 \"no error handling on the email leg\"; 52-53, 62-63, 96-97 mail client rethrows to handler; 88-89 client durably records failed attempt; 66-69 runbook retries notification only; 70-73 escaped exceptions -> 500 -> Stripe retry; 34-37, 72-73 completion marker only after clean commit. Mail-call position relative to DB transaction: **unknown**. | Mail exception escapes handler; webhook 500; Stripe retries committed payment | A) commit, rescue named mail errors, 200; B) keep as written; C) rescue MailTimeout only | unresolved | pending |", + "reconstructedGrid": [ + "Commitment | Source/approval or pending | Current (plan) | A | B | C", + "---|---|---|---|---|---", + "Mail send happens after user update commits | pending (FAN-1) | unspecified | required | unspecified | required", + "Mail exceptions rescued in handler | pending (FAN-1) | none | MailTimeout + named delivery errors | none | MailTimeout only", + "DB exceptions propagate to wrapper (500 -> retry) | PLAN.md 70-73 retained | yes | yes (unchanged) | yes | yes", + "Failed send durably recorded by mail client | PLAN.md 88-89 retained | yes | yes (relied on) | yes | yes", + "Handler outcome after mail failure | pending (FAN-1) | 500, Stripe retries | 200, completion marked, structured warning w/ event+user+PI ids | 500, Stripe retries | 200 on timeout; 500 on other mail errors", + "Receipt retry path | PLAN.md 66-69 retained | runbook + Stripe retry | runbook only (single path) | both (duplicate paths) | mixed", + "Regression tests for this behavior | pending (TEST-1 owns method) | none | required with the change | n/a | required with the change", + "Other rows (DB-1, TEST-1, PERF-1) | pending | pending | pending | pending | pending" + ] + }, + { + "attemptDir": "shards/skill-e2e-plan-ceo-finding-count/pty-count/local-1995133f-7dbe-4594-ba1d-5f835bfc38ec/plan-ceo-review-1790577677010-rYjF3v", + "observedError": "Unsupported current CEO decision; cannot exclude it from the 4–7 count: b9c36bcf-1992-4348-981e-d2f91f50e17f:toolu_012We34L2WH9RqcDKxEUsg4S", + "seedPlanPath": "/home/runner/.cache/gstack-paid-shard-DAsxIf/tmp/gstack-e2e-plan-ceo-sc4Hzf/gstack-test-plan-ceo.md", + "call": { + "sessionId": "b9c36bcf-1992-4348-981e-d2f91f50e17f", + "toolUseId": "toolu_012We34L2WH9RqcDKxEUsg4S", + "questions": [ + { + "question": "D4 — ERR-1: What happens when the receipt email fails after the payment update?\nProject/branch/task: main — payment webhook handler, HOLD SCOPE, Section 2 error map.\nELI10: The handler does two things in a row: mark the user paid in the database, then ask the mail service to send a receipt. The plan does nothing if the mail service fails, so the error bubbles up and Stripe is told \"this whole webhook failed.\" Stripe then re-sends a payment that already went through, and on-call gets paged for a payment failure that is really a mail hiccup. The mail service already saves a retry record and raises its own alert when it fails, so the handler can safely say \"payment done, receipt pending.\"\nStakes if we pick wrong: a one-hour mail outage turns into hours of Stripe retries against committed payments, a noisy payment alert, and a runbook that has to explain the difference every time.\nRecommendation: B because the retained contracts (PLAN.md:66-69, 88-91) already promise \"never replay the payment\" and a durable notification retry record; rescuing only the two named mail exceptions after commit is the smallest change that honors both, with zero catch-alls (engineering preference: name each error's class, no silent failures).\nCompleteness: A=3/10, B=10/10, C=8/10\nPros / cons: see each option.\nNet: A keeps the plan text and makes Stripe your notification retry engine; B keeps payment and notification failures on separate rails using only existing machinery; C does the same but spends a shared-ingress edit to get there.", + "header": "ERR-1", + "options": [ + { + "label": "Commit first, then rescue the two named mail exceptions and acknowledge (recommended)", + "description": "Effort S, risk low, reuse: mail client retry record, failure-rate and backlog alerts, notification retry runbook, DB/mail outcome traces. Pin the order: user update commits, then the mail call runs outside the transaction. Rescue only MailTimeout and the mail client's delivery-failure class (owner names it); emit one structured warn with event id, user id, PaymentIntent id, exception class, outcome=receipt_deferred; return success so the guard records completion and Stripe gets 200. DB exceptions still propagate to 500. Verification: unit spec \"mail raises MailTimeout after commit → user paid, retry record present, handler returns ok\"; unit spec \"DB raises during update → exception propagates, no mail call\"; unit spec \"mail raises unknown class → propagates\" (proves no catch-all). ✅ Payment and notification failures stay on their existing separate rails; no ingress change. ✅ Failure visibility is already wired: dashboard failure rate, backlog age alert, correlated traces. ❌ Relies on the mail client's \"durably records before rethrow\" contract; the owner must verify it before this rescue is safe." + }, + { + "label": "Keep as written (no rescue; mail failure → HTTP 500)", + "description": "Effort S (zero implementation work), risk high, reuse: none. ✅ No new code in the handler; Stripe's backoff acts as a free retry when the completion marker is not recorded on exception. ✅ Nothing to get wrong in a rescue clause. ❌ A committed payment answers 500; the failed-webhook alert pages for mail outages and the runbook contract \"never replay the payment\" is broken by design. ❌ Behavior depends on an unverified guard detail (marker recorded or not after a post-commit raise)." + }, + { + "label": "Rescue and re-raise as a handler-level NotificationFailed that ingress maps to 200", + "description": "Effort M, risk medium, reuse: same as B plus ingress wrapper. Same commit-then-send order and the same named rescues, but wrap into a new exception the ingress wrapper recognizes and converts to a 200 with outcome=notification_failed. Verification: B's specs plus an ingress spec for the mapping. ✅ Ingress logs carry an explicit notification-failed outcome without the handler swallowing anything. ✅ Same separation of payment vs notification failure as B. ❌ Touches the shared ingress wrapper other handlers depend on, adding a second policy location for one handler's need. ❌ More surface for the same user-visible result as B." + } + ], + "multiSelect": false + } + ], + "answered": true, + "failed": false, + "answers": { + "D4 — ERR-1: What happens when the receipt email fails after the payment update?\nProject/branch/task: main — payment webhook handler, HOLD SCOPE, Section 2 error map.\nELI10: The handler does two things in a row: mark the user paid in the database, then ask the mail service to send a receipt. The plan does nothing if the mail service fails, so the error bubbles up and Stripe is told \"this whole webhook failed.\" Stripe then re-sends a payment that already went through, and on-call gets paged for a payment failure that is really a mail hiccup. The mail service already saves a retry record and raises its own alert when it fails, so the handler can safely say \"payment done, receipt pending.\"\nStakes if we pick wrong: a one-hour mail outage turns into hours of Stripe retries against committed payments, a noisy payment alert, and a runbook that has to explain the difference every time.\nRecommendation: B because the retained contracts (PLAN.md:66-69, 88-91) already promise \"never replay the payment\" and a durable notification retry record; rescuing only the two named mail exceptions after commit is the smallest change that honors both, with zero catch-alls (engineering preference: name each error's class, no silent failures).\nCompleteness: A=3/10, B=10/10, C=8/10\nPros / cons: see each option.\nNet: A keeps the plan text and makes Stripe your notification retry engine; B keeps payment and notification failures on separate rails using only existing machinery; C does the same but spends a shared-ingress edit to get there.": "Commit first, then rescue the two named mail exceptions and acknowledge (recommended)" + }, + "unansweredQuestionIndices": [], + "answeredAt": "2026-09-28T06:48:52.629Z" + }, + "reconstructedLedgerRow": "| ERR-1 (Section 2) | PLAN.md:52-53, 60-63, 88-97, 114-116: mail client rethrows; plan has no error handling on email leg; DB exceptions → 500 → Stripe retry | Email exception propagates to ingress wrapper → HTTP 500 | pending (Section 2) | unresolved | pending |" + } + ] +} \ No newline at end of file diff --git a/test/helpers/ceo-payment-findings.ts b/test/helpers/ceo-payment-findings.ts index b3ce086d9..a38861415 100644 --- a/test/helpers/ceo-payment-findings.ts +++ b/test/helpers/ceo-payment-findings.ts @@ -278,7 +278,7 @@ function attributedBaselineDefect(q: NativePlanQuestionCall['questions'][number] /** Source requires Current/Proposed/Status/evidence and a cited row ID. It * does not require heading depth, column order, a Dn(ledger ID) title, or * native option wording. Pending is valid: the actual ACK precedes the next Edit. */ -export function ceoPaymentFinding(fp: AskUserQuestionFingerprint, seedPlan: string, savedPlan: string): Finding | null { +export function ceoPaymentFinding(fp: AskUserQuestionFingerprint, seedPlan: string, savedPlan: string, trace?: string[]): Finding | null { const call = ownedAnswer(fp); if (!call) return null; const q = call.questions[0]!; @@ -334,14 +334,22 @@ export function ceoPaymentFinding(fp: AskUserQuestionFingerprint, seedPlan: stri excludesCurrentTestTarget(cells[fields.current[0]!]!.text) && excludesCurrentTestTarget(defectExplanation); const pendingDefect = /^(?:unresolved|reopened)\b/i.test(status) && current(read('proposed')) && spec.subject.test(read('proposed')) && spec.defect.test(defectValue('proposed')); - if (!spec.subject.test(seedPlan) || !spec.defect.test(seedPlan) || !spec.subject.test(row) || - !(spec.defect.test(defectValue('current')) || pendingDefect || scopedTestAbsence) || - !spec.subject.test(question) || !(spec.defect.test(defectExplanation) || scopedTestAbsence || + const checks = { + seedPlan: spec.subject.test(seedPlan) && spec.defect.test(seedPlan), + rowSubject: spec.subject.test(row), + rowDefect: spec.defect.test(defectValue('current')) || pendingDefect || scopedTestAbsence, + questionSubject: spec.subject.test(question), + explanationDefect: spec.defect.test(defectExplanation) || scopedTestAbsence || (pendingDefect && declaredSources.length <= 1 && declaredSources.every(source => source === 'PLAN.md') && - currentDocumentContext(tokens, tokens.indexOf(table)) && attributedBaselineDefect(q, read('proposed'), explanation, spec)))) continue; - const operative = options.some(o => spec.remedy.test(o) && spec.subject.test(o)); + currentDocumentContext(tokens, tokens.indexOf(table)) && attributedBaselineDefect(q, read('proposed'), explanation, spec)), + operativeOption: options.some(o => spec.remedy.test(o) && spec.subject.test(o)), + proposal: proposals.some(p => (!(scopedTestAbsence || pending) || p.active) && current(p.body) && spec.remedy.test(p.body) && spec.subject.test(p.body)), + }; + if (trace && (checks.rowSubject || checks.questionSubject)) + trace.push(`${spec.seed}@${id}: ${Object.entries(checks).map(([name, ok]) => `${name}=${ok ? 'yes' : 'no'}`).join(' ')}`); + if (!checks.seedPlan || !checks.rowSubject || !checks.rowDefect || !checks.questionSubject || !checks.explanationDefect) continue; const proposal = proposals.find(p => (!(scopedTestAbsence || pending) || p.active) && current(p.body) && spec.remedy.test(p.body) && spec.subject.test(p.body)); - if (operative && proposal) matches.push({ seed: spec.seed, ledgerId: id, phase: proposal.phase, signature: fp.signature }); + if (checks.operativeOption && proposal) matches.push({ seed: spec.seed, ledgerId: id, phase: proposal.phase, signature: fp.signature }); } } } @@ -961,7 +969,11 @@ export function createCeoPaymentFindingCounter(seedPlan: string, readPlan: () => if (decision) { trace.push({ signature: fp.signature, kind: 'recorded-decision', ...decision }); return true; } if (todoDecision(fp)) { trace.push({ signature: fp.signature, kind: 'additional-current-decision' }); return true; } if (existingFinding(fp)) { trace.push({ signature: fp.signature, kind: 'existing-finding' }); return true; } - throw new Error(`Unsupported current CEO decision; cannot exclude it from the 4–7 count: ${fp.signature}`); + const q = fp.nativeCall!.questions[0]!, predicates: string[] = []; + ceoPaymentFinding(fp, seedPlan, plan, predicates); + throw new Error(`Unsupported current CEO decision; cannot exclude it from the 4–7 count: ${fp.signature}\n` + + `header: ${q.header}\nquestion: ${q.question.slice(0, 200)}\n` + + `obligation predicates: ${predicates.length ? predicates.join('; ') : 'no active PLAN.md-sourced ledger row named by the question shares an obligation subject'}`); }, }; }