fix(test): recognize ledger row-ID split candidates so collection stops at the last ACK

Run 36385945043's split-overflow case asked all five candidate decisions by
8m55s, but the live candidate check required the question to open with
"E1:" and every option to be a known disposition. The skill cited ledger
row IDs ("D2.1 — R-E1: …") and offered "Hold, discuss first", so no
candidate was recognized and the attempt ran the whole review (1302s).

Identity now comes from the native header; the question must open with that
candidate's ledger reference, name only that candidate, and offer exactly one
include, defer and cut disposition. The selected answer must still be one of
those three. The semantic evaluator and every existing negative control are
unchanged; a trimmed capture from the run adds the positive case and four
row-ID negative controls.
This commit is contained in:
garrytan committed 2026-09-29 15:18:46 +00:00
1 parent 2ba451a49d
commit dc5934ee54
3 files changed
+430 -11

No files matched your search

+38 -1
View File
@@ -4,8 +4,9 @@ import * as os from 'node:os';
import * as path from 'node:path';
import { nativePlanCallFingerprint } from './helpers/claude-pty-runner';
import type { NativePlanQuestionCall, PlanCountTranscript } from './helpers/plan-count-transcript';
import { ceoSplitCandidate, ceoSplitDecisionFingerprints, isCeoSplitCollectionComplete } from './helpers/ceo-split-question-policy';
import { ceoSplitCandidate, ceoSplitDecisionFingerprints, isCeoSplitCandidateCall, isCeoSplitCollectionComplete } from './helpers/ceo-split-question-policy';
import captured from './fixtures/ceo-split-collection-0bcd.json';
import rowIds from './fixtures/ceo-split-collection-3638.json';
const ROOT = path.resolve(import.meta.dir, '..');
function original() {
@@ -50,6 +51,42 @@ test.each([0, 1, 2, 3, 4, 5, 6])('the exact original %i-call prefix waits for th
expect(accepts(state)).toBe(length === 6);
});
// Run 36385945043: the skill cited ledger row IDs ("D2.1 — R-E1: …") and offered
// a fourth "Hold, discuss first" option. No candidate was recognized, so collection
// never stopped and the attempt ran the whole review (1302s) after the E5 ACK.
function rowIdCapture() {
const calls = structuredClone(rowIds.calls) as NativePlanQuestionCall[];
const transcript: PlanCountTranscript = { status: 'ready', calls, assistantMessages: [] };
const fingerprints = rowIds.fingerprints.map((fp, index) => ({ ...structuredClone(fp), nativeCall: calls[index]! }));
return { transcript, fingerprints };
}
test('ledger row-ID candidate questions from run 36385945043 finish collection at the E5 ACK', () => {
const state = rowIdCapture();
expect(rowIds.provenance.originalOutcome).toBe('completion_summary');
expect(rowIds.provenance.originalReviewCount).toBe(0);
expect(state.transcript.calls.at(-1)!.answeredAt).toBe(rowIds.provenance.completeAt);
expect(state.transcript.calls.map(call => ceoSplitCandidate(call.questions[0]!))).toEqual([null, 'E1', 'E2', 'E3', 'E4', 'E5']);
expect(state.fingerprints.map(isCeoSplitCandidateCall)).toEqual([false, true, true, true, true, true]);
for (let length = 0; length < 6; length++) {
const prefix = rowIdCapture();
prefix.transcript.calls.length = length; prefix.fingerprints.length = length;
expect(accepts(prefix)).toBe(false);
}
expect(accepts(state)).toBe(true);
});
test.each(['foreign_row', 'quoted_row', 'second_platform', 'held'])('row-ID collection rejects %s evidence', kind => {
const state = rowIdCapture(), call = state.transcript.calls.at(-1)!, question = call.questions[0]!;
const selected = call.answers![question.question]!;
if (kind === 'foreign_row') question.question = question.question.replace('R-E5:', 'R-E4:');
if (kind === 'quoted_row') question.question = 'Example: ' + question.question;
if (kind === 'second_platform') question.question = question.question.replace('?', ' or the Slack bot?');
call.answers = { [question.question]: kind === 'held' ? question.options[3]!.label : selected };
state.fingerprints = fromCalls(state.transcript.calls).fingerprints;
expect(accepts(state)).toBe(false);
});
test('four candidate calls with five independent tabs meet the original floor', () => {
const state = grouped(4);
expect(state.transcript.calls).toHaveLength(5); // Four candidate calls plus mode.