diff --git a/CHANGELOG.md b/CHANGELOG.md index 9c2f4861f..a6847d87a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -32,11 +32,13 @@ Run `bun run typecheck` and `bun run typecheck:test` before you push; both are f - Compiled `/cso` installs called an unimported `join` when launching the assertion-witness child, breaking runtime-tested witnessing for every installed user. - `/review` workflow ambiguities (smoke clock vs required revalidation, setup authority, plan-completion gate, findings record), `/office-hours` builder mode not loading its brainstorm section, `/sync-gbrain` Step 4 helper arguments and write path, `/plan-ceo-review` expansion framing and pacing menus, `/plan-design-review` with no designer API key, and `/deslop-shared-libs` one-file-per-turn reads. - Eval detectors that graded wording or step order now grade outcomes: eng batching, CEO split-overflow, mode routing, section-loading stale-fill, outside-voice-disabled attribution, design focus menus, and PTY permission dialogs with cropped titles. +- Harness races and adapter gaps found by the proof runs: plan seeding accepted a stale empty input box when the CLI repainted after recording its reply, the third-party-actions recorder fixture lost every failure record, the autoplan dual-voice check could not read framed subagent reports from newer Claude Code, and the HOLD SCOPE routing check judged the skill's own defer/keep menu as its rigor decision, the outside-disabled check missed a correctly attributed quote of the pre-existing review record, and the plan-review judge was not told its reason length bound on the field it writes. #### Changed - Paid evals: one test file or case per machine within a 540-second slice budget, planned from recorded per-tier and per-case durations; case sharding for plan, design, review-army, shared-libs, shared-libs-paths, ship-docsync and qa-callers. - Verdict policy: no retries; `rule` cases fail on any failed trial, `behavior` cases pass on 2 of 3 parallel trials with contract assertions still strict, `judge` entries average 3 samples against unchanged thresholds. One panel-verdict function feeds the report, PR comment, weekly issue and pass-rate history. A census whose every red is infrastructure is re-dispatched once, and both runs are reported. -- New non-blocking weekly `evals-marathon.yml` lane for full start-to-finish flows (the full `/office-hours` workflow; a focused design-draft case replaces it in the weekly lane). +- New non-blocking weekly `evals-marathon.yml` lane for full start-to-finish flows: the full `/office-hours` workflow (a focused design-draft case replaces it in the weekly lane) and the full `/plan-ceo-review` split-overflow run, which took 8 to 20 minutes on its own. +- The CI image stays on Claude Code 2.1.251. On 2.1.284, sessions ran 20% longer on the same model (per-turn effort) and 11 gate cases timed out on unchanged budgets; that bump needs its own budget work. - `lib/cso/*.ts` is formatted with pinned Prettier; minified transpile output is byte-identical except three canonicalized regex flag orders. - The duplicate dispatch-only `ship-docsync` case is removed; `ship-docsync-completion` asserts the same on the same fixture. diff --git a/test/disabled-dated-record-at.test.ts b/test/disabled-dated-record-at.test.ts index 565fa9d91..44aa0dc6e 100644 --- a/test/disabled-dated-record-at.test.ts +++ b/test/disabled-dated-record-at.test.ts @@ -247,3 +247,24 @@ test('b176 retained history cannot override actual current log completion or nat expect(disabledPlanReviewEvidence(item.result, 'codex invoked\n', item.reviewLog, item.priorRecord).passed).toBe(false); expect(disabledPlanReviewEvidence(sourceBoundB176.cases[1]!.result, '', sourceBoundB176.cases[1]!.reviewLog, item.priorRecord).passed).toBe(false); }); + +describe('parenthesized field list of the retained record (census 36629958451)', () => { + const prior = fixture.cases[0]!.priorRecord as Record; + const list = (items: string[]) => `One thing worth flagging: the fixture's review log already contained a prior entry (${items.map(item => '`' + item + '`').join(', ')}) that predates this run. I did not produce it and it does not count toward this review's coverage.`; + const exact = [prior.timestamp!, `status: ${prior.status}`, `source: ${prior.source}`, 'outside_status: completed']; + test('the exact timestamp and the record\'s own field values attribute the quotation to that record', () => { + expect(evaluate(0, list(exact))).toMatchObject({ passed: true, falseCompletion: false }); + }); + test.each([ + ['a different timestamp', [prior.timestamp!.replace(/\d(?=Z$)/, d => String((Number(d) + 1) % 10)), ...exact.slice(1)]], + ['no timestamp', exact.slice(1)], + ['a field value the record lacks', [prior.timestamp!, 'status: completed', `source: ${prior.source}`, 'outside_status: completed']], + ['free prose inside the list', [prior.timestamp!, `status: ${prior.status}`, 'outside_status: completed and this run too']], + ])('%s is not attributed', (_name, items) => { + expect(evaluate(0, list(items as string[])).falseCompletion).toBe(true); + }); + test('a current mutation or a separate current claim still fails', () => { + expect(evaluate(0, list(exact).replace('already contained', 'now appended')).falseCompletion).toBe(true); + expect(evaluate(0, `${list(exact)}\nThis run: outside_status: completed.`).falseCompletion).toBe(true); + }); +}); diff --git a/test/helpers/disabled-plan-review-fixture.ts b/test/helpers/disabled-plan-review-fixture.ts index 884d5525a..a9e49ab90 100644 --- a/test/helpers/disabled-plan-review-fixture.ts +++ b/test/helpers/disabled-plan-review-fixture.ts @@ -173,6 +173,24 @@ function withoutAttributedPriorRecordData(output: string, priorRecord?: Record item.replace(/[*`]/g, '').trim()); + const fieldsOk = items.every(item => { + if (item === priorRecord.timestamp) return true; + const field = /^["']?([a-z_]+)["']?\s*[:=]\s*["']?([a-z0-9_.:+-]+)["']?$/i.exec(item); + return !!field && fields.has(field[1]!) && field[1] !== 'timestamp' && priorRecord[field[1]!] === field[2]; + }); + if (!fieldsOk || !items.includes(String(priorRecord.timestamp)) || !items.some(item => /^["']?outside_status\b/i.test(item))) continue; + const start = match.index + match[0].length - match[1]!.length - 1; + spans.push({ start, end: start + match[1]!.length + 2 }); + } for (const span of spans.sort((a, b) => b.start - a.start)) { output = output.slice(0, span.start) + output.slice(span.start, span.end).replace(/[^\r\n]/g, ' ') + output.slice(span.end); } diff --git a/test/helpers/plan-review-decisions.ts b/test/helpers/plan-review-decisions.ts index c8234459c..d53f1a5f0 100644 --- a/test/helpers/plan-review-decisions.ts +++ b/test/helpers/plan-review-decisions.ts @@ -54,6 +54,9 @@ export interface PlanReviewDecisionJudgment { engReview?: EngReviewJudgment; } export type PlanReviewJudge = (prompt: string, model?: string, opts?: Pick) => Promise; +// Structured outputs cannot enforce maxLength, so the reason bound the local +// validator applies is stated on the field the model writes. +const REASON_FIELD = { type: 'string', description: '1-1000 characters: under 120 words.' } as const; // Only response structure is constrained. Identity, exact quotes, enum casing, // uncertainty, target coverage, independence and count checks remain local. function planReviewDecisionSchema(withPeerComparison: boolean, withEngReview = false): NonNullable { @@ -68,7 +71,7 @@ function planReviewDecisionSchema(withPeerComparison: boolean, withEngReview = f toolUseId: { type: 'string' }, questionIndex: { type: 'integer' }, kind: { type: 'string', enum: ['finding', 'scope', 'workflow', 'backlog', 'uncertain'] }, targetIds: { type: 'array', items: { type: 'string' } }, - independentDecisions: { type: 'integer' }, reason: { type: 'string' }, + independentDecisions: { type: 'integer' }, reason: REASON_FIELD, evidence: { type: 'array', items: { type: 'object', additionalProperties: false, required: ['field', 'optionIndex', 'quote'], properties: { @@ -86,7 +89,7 @@ function planReviewDecisionSchema(withPeerComparison: boolean, withEngReview = f type: 'object', additionalProperties: false, required: ['status', 'regression', 'approvals', 'navigation', 'reason'], properties: { - status: { type: 'string', enum: ['complete', 'missing', 'uncertain'] }, reason: { type: 'string' }, + status: { type: 'string', enum: ['complete', 'missing', 'uncertain'] }, reason: REASON_FIELD, regression: { type: 'array', items: { type: 'object', additionalProperties: false, required: ['role', 'source', 'quote'], properties: { role: { type: 'string', enum: ['critical', 'baseline', 'replay', 'assertions', 'approved-differences'] }, @@ -112,7 +115,7 @@ function planReviewDecisionSchema(withPeerComparison: boolean, withEngReview = f properties: { name: { type: 'string' }, quote: { type: 'string' } }, } }, productQuote: { type: 'string' }, groundingQuote: { type: 'string' }, - implicationQuote: { type: 'string' }, reason: { type: 'string' }, + implicationQuote: { type: 'string' }, reason: REASON_FIELD, }, } } : {}), }, diff --git a/test/helpers/touchfiles-data.ts b/test/helpers/touchfiles-data.ts index 4af7e14d6..35ccaf4e0 100644 --- a/test/helpers/touchfiles-data.ts +++ b/test/helpers/touchfiles-data.ts @@ -1276,7 +1276,7 @@ export const E2E_TIERS: Record = { 'plan-design-finding-floor': 'periodic', // stochastic ask-first (see plan-mode-handshake note); periodic 'plan-devex-finding-floor': 'gate', 'plan-eng-multi-finding-batching': 'periodic', - 'plan-ceo-split-overflow': 'periodic', + 'plan-ceo-split-overflow': 'marathon', // Full /plan-ceo-review through split overflow (504–1188 s on 2.1.251) // Privacy gate for gstack-brain-sync — periodic (non-deterministic LLM call, // costs ~$0.30-$0.50 per run, not needed on every commit) diff --git a/test/skill-e2e-plan-ceo-split-overflow.test.ts b/test/skill-e2e-plan-ceo-split-overflow.test.ts index d7f81222f..392cef32d 100644 --- a/test/skill-e2e-plan-ceo-split-overflow.test.ts +++ b/test/skill-e2e-plan-ceo-split-overflow.test.ts @@ -1,5 +1,5 @@ /** - * /plan-ceo-review split-overflow regression (periodic, paid, real-PTY). + * /plan-ceo-review split-overflow regression (marathon, paid, real-PTY). * * Catches the original failure mode the user complained about: when the * agent has 5+ options for ONE conceptual decision, it must split into N @@ -50,7 +50,7 @@ import { ceoSplitDecisionFingerprints, isCeoSplitCandidateCall, isCeoSplitCollec import { CEO_SCOPE_CANDIDATES } from './helpers/plan-review-cases'; import { evaluatePlanReviewDecisions } from './helpers/plan-review-decisions'; -const describeE2E = describeE2ETier('periodic'); +const describeE2E = describeE2ETier('marathon'); const N = 5; const FLOOR = N - 1; // 4 — must fire at least one AUQ per non-dropped option @@ -60,7 +60,7 @@ const FLOOR = N - 1; // 4 — must fire at least one AUQ per non-dropped option * EVALS_JOBS>1, sibling worktrees) never share one /tmp artifact. */ const FIXTURE_PLAN_PATH = '/tmp/gstack-test-plan-ceo-split-overflow.md'; -describeE2E('/plan-ceo-review split-overflow regression (periodic)', () => { +describeE2E('/plan-ceo-review split-overflow regression (marathon)', () => { test( `5-option scope decision emits >= ${FLOOR} review-phase AskUserQuestions (no dropping)`, async () => {