test: attribute quoted prior-record field lists, state the judge reason bound in its schema, move split-overflow to marathon

Census 36629958451 reds:
- outside-plan-disabled-no-fallback: the model quoted the pre-existing record
  as a parenthesized field list with its exact timestamp; attribution now
  requires that exact timestamp and the record's own field values.
- plan-devex-peer-comparison-classification: the judge correctly returned
  missing but wrote a 1069-character reason, voiding the judgment; structured
  outputs cannot enforce maxLength, so the bound is stated on the field.
- plan-ceo-split-overflow ran 504-1188 s as one PTY flow and set the
  periodic lane's wall clock; it now runs weekly in the marathon lane.
This commit is contained in:
garrytan committed 2026-09-29 21:26:37 +00:00
1 parent f63e1fb7cb
commit bf4667146c
6 files changed
+52 -8

No files matched your search

+21
View File
@@ -247,3 +247,24 @@ test('b176 retained history cannot override actual current log completion or nat
expect(disabledPlanReviewEvidence(item.result, 'codex invoked\n', item.reviewLog, item.priorRecord).passed).toBe(false);
expect(disabledPlanReviewEvidence(sourceBoundB176.cases[1]!.result, '', sourceBoundB176.cases[1]!.reviewLog, item.priorRecord).passed).toBe(false);
});
describe('parenthesized field list of the retained record (census 36629958451)', () => {
const prior = fixture.cases[0]!.priorRecord as Record<string, string>;
const list = (items: string[]) => `One thing worth flagging: the fixture's review log already contained a prior entry (${items.map(item => '`' + item + '`').join(', ')}) that predates this run. I did not produce it and it does not count toward this review's coverage.`;
const exact = [prior.timestamp!, `status: ${prior.status}`, `source: ${prior.source}`, 'outside_status: completed'];
test('the exact timestamp and the record\'s own field values attribute the quotation to that record', () => {
expect(evaluate(0, list(exact))).toMatchObject({ passed: true, falseCompletion: false });
});
test.each([
['a different timestamp', [prior.timestamp!.replace(/\d(?=Z$)/, d => String((Number(d) + 1) % 10)), ...exact.slice(1)]],
['no timestamp', exact.slice(1)],
['a field value the record lacks', [prior.timestamp!, 'status: completed', `source: ${prior.source}`, 'outside_status: completed']],
['free prose inside the list', [prior.timestamp!, `status: ${prior.status}`, 'outside_status: completed and this run too']],
])('%s is not attributed', (_name, items) => {
expect(evaluate(0, list(items as string[])).falseCompletion).toBe(true);
});
test('a current mutation or a separate current claim still fails', () => {
expect(evaluate(0, list(exact).replace('already contained', 'now appended')).falseCompletion).toBe(true);
expect(evaluate(0, `${list(exact)}\nThis run: outside_status: completed.`).falseCompletion).toBe(true);
});
});
@@ -173,6 +173,24 @@ function withoutAttributedPriorRecordData(output: string, priorRecord?: Record<s
spans.push({ start: match.index, end: match.index + match[0].length });
}
}
// A parenthesized field list right after a pre-existing-record owner is that
// record when it quotes the record's exact ISO timestamp and every other item
// is one of its own field values; a current mutation before the owner fails.
const owned = /\b(?:earlier|prior|previous|historical|old(?:er)?|pre[- ]existing)\s+(?:(?:review[- ]log|review|log)\s+)?(?:entry|record|line|row)\s*\(([^()\r\n]+)\)/gi;
for (const match of output.matchAll(owned)) {
const lineStart = output.lastIndexOf('\n', match.index) + 1;
const local = output.slice(lineStart, match.index).split(/(?<=[.!?;])\s+/).at(-1) ?? '';
if (/\b(?:now|currently|current|today|new|updat\w*|append\w*|chang\w*|mark\w*|set|write|wrote)\b/i.test(local.replace(/[*`]/g, ''))) continue;
const items = match[1]!.split(',').map(item => item.replace(/[*`]/g, '').trim());
const fieldsOk = items.every(item => {
if (item === priorRecord.timestamp) return true;
const field = /^["']?([a-z_]+)["']?\s*[:=]\s*["']?([a-z0-9_.:+-]+)["']?$/i.exec(item);
return !!field && fields.has(field[1]!) && field[1] !== 'timestamp' && priorRecord[field[1]!] === field[2];
});
if (!fieldsOk || !items.includes(String(priorRecord.timestamp)) || !items.some(item => /^["']?outside_status\b/i.test(item))) continue;
const start = match.index + match[0].length - match[1]!.length - 1;
spans.push({ start, end: start + match[1]!.length + 2 });
}
for (const span of spans.sort((a, b) => b.start - a.start)) {
output = output.slice(0, span.start) + output.slice(span.start, span.end).replace(/[^\r\n]/g, ' ') + output.slice(span.end);
}
+6 -3
View File
@@ -54,6 +54,9 @@ export interface PlanReviewDecisionJudgment {
engReview?: EngReviewJudgment;
}
export type PlanReviewJudge = (prompt: string, model?: string, opts?: Pick<CallJudgeOptions, 'signal' | 'max_tokens' | 'jsonSchema'>) => Promise<unknown>;
// Structured outputs cannot enforce maxLength, so the reason bound the local
// validator applies is stated on the field the model writes.
const REASON_FIELD = { type: 'string', description: '1-1000 characters: under 120 words.' } as const;
// Only response structure is constrained. Identity, exact quotes, enum casing,
// uncertainty, target coverage, independence and count checks remain local.
function planReviewDecisionSchema(withPeerComparison: boolean, withEngReview = false): NonNullable<CallJudgeOptions['jsonSchema']> {
@@ -68,7 +71,7 @@ function planReviewDecisionSchema(withPeerComparison: boolean, withEngReview = f
toolUseId: { type: 'string' }, questionIndex: { type: 'integer' },
kind: { type: 'string', enum: ['finding', 'scope', 'workflow', 'backlog', 'uncertain'] },
targetIds: { type: 'array', items: { type: 'string' } },
independentDecisions: { type: 'integer' }, reason: { type: 'string' },
independentDecisions: { type: 'integer' }, reason: REASON_FIELD,
evidence: { type: 'array', items: {
type: 'object', additionalProperties: false, required: ['field', 'optionIndex', 'quote'],
properties: {
@@ -86,7 +89,7 @@ function planReviewDecisionSchema(withPeerComparison: boolean, withEngReview = f
type: 'object', additionalProperties: false,
required: ['status', 'regression', 'approvals', 'navigation', 'reason'],
properties: {
status: { type: 'string', enum: ['complete', 'missing', 'uncertain'] }, reason: { type: 'string' },
status: { type: 'string', enum: ['complete', 'missing', 'uncertain'] }, reason: REASON_FIELD,
regression: { type: 'array', items: { type: 'object', additionalProperties: false,
required: ['role', 'source', 'quote'], properties: {
role: { type: 'string', enum: ['critical', 'baseline', 'replay', 'assertions', 'approved-differences'] },
@@ -112,7 +115,7 @@ function planReviewDecisionSchema(withPeerComparison: boolean, withEngReview = f
properties: { name: { type: 'string' }, quote: { type: 'string' } },
} },
productQuote: { type: 'string' }, groundingQuote: { type: 'string' },
implicationQuote: { type: 'string' }, reason: { type: 'string' },
implicationQuote: { type: 'string' }, reason: REASON_FIELD,
},
} } : {}),
},
+1 -1
View File
@@ -1276,7 +1276,7 @@ export const E2E_TIERS: Record<string, 'gate' | 'periodic' | 'marathon'> = {
'plan-design-finding-floor': 'periodic', // stochastic ask-first (see plan-mode-handshake note); periodic
'plan-devex-finding-floor': 'gate',
'plan-eng-multi-finding-batching': 'periodic',
'plan-ceo-split-overflow': 'periodic',
'plan-ceo-split-overflow': 'marathon', // Full /plan-ceo-review through split overflow (504–1188 s on 2.1.251)
// Privacy gate for gstack-brain-sync — periodic (non-deterministic LLM call,
// costs ~$0.30-$0.50 per run, not needed on every commit)
@@ -1,5 +1,5 @@
/**
* /plan-ceo-review split-overflow regression (periodic, paid, real-PTY).
* /plan-ceo-review split-overflow regression (marathon, paid, real-PTY).
*
* Catches the original failure mode the user complained about: when the
* agent has 5+ options for ONE conceptual decision, it must split into N
@@ -50,7 +50,7 @@ import { ceoSplitDecisionFingerprints, isCeoSplitCandidateCall, isCeoSplitCollec
import { CEO_SCOPE_CANDIDATES } from './helpers/plan-review-cases';
import { evaluatePlanReviewDecisions } from './helpers/plan-review-decisions';
const describeE2E = describeE2ETier('periodic');
const describeE2E = describeE2ETier('marathon');
const N = 5;
const FLOOR = N - 1; // 4 — must fire at least one AUQ per non-dropped option
@@ -60,7 +60,7 @@ const FLOOR = N - 1; // 4 — must fire at least one AUQ per non-dropped option
* EVALS_JOBS>1, sibling worktrees) never share one /tmp artifact. */
const FIXTURE_PLAN_PATH = '/tmp/gstack-test-plan-ceo-split-overflow.md';
describeE2E('/plan-ceo-review split-overflow regression (periodic)', () => {
describeE2E('/plan-ceo-review split-overflow regression (marathon)', () => {
test(
`5-option scope decision emits >= ${FLOOR} review-phase AskUserQuestions (no dropping)`,
async () => {