mirror of
https://github.com/garrytan/gstack.git
synced 2026-10-02 17:40:02 +02:00
ci(image): keep Claude Code 2.1.251; test(ceo-mode-routing): keep HOLD's own deferrals in scope before assessing its rigor decision
2.1.284 enables per-turn effort for claude-fable-5-1: in gate census 36626737820, 66 of 84 sessions ran longer than on 2.1.251 (+20% session time, +32% thinking tokens) and 11 cases timed out on unchanged budgets. HOLD SCOPE's 0G step asks its own defer/keep menu; the actor answered it Defer and the assessment then judged that scope question as the rigor decision. The actor now answers that menu Keep and assesses the next one.
This commit is contained in:
1 parent
d05d161713
commit
f63e1fb7cb
5 files changed
+55
-13
No files matched your search
@@ -93,13 +93,12 @@ RUN curl --retry 5 --retry-delay 5 --retry-connrefused -fsSL https://bun.sh/inst
|
||||
# skillify HOME discovery on 2.1.237, guard/freeze hooks on 2.1.162).
|
||||
# Bump deliberately, via a PR that runs the PTY gate against the new TUI.
|
||||
# test/ci-image-cli-pin.test.ts fails the free suite if this pin is removed.
|
||||
# 2.1.284 (from 2.1.251, 2026-09-29): 2.1.251 logs
|
||||
# [claude-code:unrecognized_model] for the eval model claude-fable-5-1;
|
||||
# 2.1.284 recognizes it. Local canary: the gate PTY smoke subset parsed on
|
||||
# both TUIs (7/8 pass on 2.1.284, 8/8 on 2.1.251; the one red was the model
|
||||
# still in WebSearch at 300 s), and plan-design-review-plan-mode passed at
|
||||
# 293 s on 2.1.284 where 2.1.251 timed out at 300 s.
|
||||
RUN npm i -g @anthropic-ai/claude-code@2.1.284
|
||||
# Stays on 2.1.251. 2.1.284 (tried 2026-09-29) enables per-turn effort for
|
||||
# claude-fable-5-1: in gate census run 36626737820, 66 of 84 sessions ran
|
||||
# longer than the same cases in run 36606688266 on 2.1.251 (session time
|
||||
# +20%, thinking tokens +32%), and 11 cases timed out on unchanged budgets.
|
||||
# Bumping it needs its own budget and skill-speed work.
|
||||
RUN npm i -g @anthropic-ai/claude-code@2.1.251
|
||||
|
||||
# Playwright system deps (Chromium) — needed for browse E2E tests
|
||||
RUN npx playwright install-deps chromium
|
||||
|
||||
@@ -30,7 +30,6 @@ Run `bun run typecheck` and `bun run typecheck:test` before you push; both are f
|
||||
#### Fixed
|
||||
- `$B connect --supervise` respawned with a block-scoped env that no longer existed, so every restart threw and the supervisor gave up after five tries. The headed env is one helper used by connect and respawn, and the loop has behavioral tests.
|
||||
- Compiled `/cso` installs called an unimported `join` when launching the assertion-witness child, breaking runtime-tested witnessing for every installed user.
|
||||
- The CI image pins Claude Code 2.1.284, which recognizes the default eval model; 2.1.251 logged `unrecognized_model` and stalled mid-stream.
|
||||
- `/review` workflow ambiguities (smoke clock vs required revalidation, setup authority, plan-completion gate, findings record), `/office-hours` builder mode not loading its brainstorm section, `/sync-gbrain` Step 4 helper arguments and write path, `/plan-ceo-review` expansion framing and pacing menus, `/plan-design-review` with no designer API key, and `/deslop-shared-libs` one-file-per-turn reads.
|
||||
- Eval detectors that graded wording or step order now grade outcomes: eng batching, CEO split-overflow, mode routing, section-loading stale-fill, outside-voice-disabled attribution, design focus menus, and PTY permission dialogs with cropped titles.
|
||||
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
import { describe, expect, test } from 'bun:test';
|
||||
import { findCeoModeOption, hasPostAnswerCeoPosture, hasNativePostAnswerCeoPosture, nativeCeoModeAnswer, nextCeoModeNavigation, nextCeoPostureContinuation } from './helpers/ceo-mode-option';
|
||||
import { findCeoModeOption, hasPostAnswerCeoPosture, hasNativePostAnswerCeoPosture, holdDeferKeepIndex, nativeCeoModeAnswer, nextCeoModeNavigation, nextCeoPostureContinuation } from './helpers/ceo-mode-option';
|
||||
import { parseNumberedOptions, stripAnsi, planCountQuestionInput, nativePlanCallFingerprint } from './helpers/claude-pty-runner';
|
||||
import type { PlanCountTranscript } from './helpers/plan-count-transcript';
|
||||
import * as fs from 'node:fs';
|
||||
@@ -1893,3 +1893,31 @@ describe('mode submission when the review panel scrolls past the viewport', () =
|
||||
expect(scrolledSubmit(scrolledReview.screen, scrolledReview.screenText, 'HOLD SCOPE', other)).toBeNull();
|
||||
});
|
||||
});
|
||||
|
||||
describe('HOLD SCOPE defer/keep menu (census 36626737820: "Defer update to TODOS.md" was answered as the rigor decision)', () => {
|
||||
const call = (labels: string[], extra: Record<string, unknown> = {}) => ({
|
||||
sessionId: 's', toolUseId: 't', answered: false, failed: false,
|
||||
questions: [{ question: 'D4 — R1: Defer the update endpoint (rename / overwrite a saved view) or keep it in scope?', header: 'Scope', multiSelect: false,
|
||||
options: labels.map(label => ({ label, description: 'd' })) }], ...extra,
|
||||
}) as any;
|
||||
test.each([
|
||||
[['Defer update to TODOS.md', 'Keep update in scope'], 2],
|
||||
[['A) Defer this item to TODOS.md', 'B) Keep it in scope (recommended)'], 2],
|
||||
[['Keep it in scope', 'Defer this item to TODOS'], 1],
|
||||
])('keeps the item in scope: %j', (labels, index) => expect(holdDeferKeepIndex(call(labels))).toBe(index));
|
||||
test.each([
|
||||
['a rigor remedy', ['Add a 404 contract test', 'Leave the criterion untested']],
|
||||
['a third option', ['Defer update to TODOS.md', 'Keep update in scope', 'Cut update']],
|
||||
['a cut instead of a deferral', ['Cut update from the plan', 'Keep update in scope']],
|
||||
['keep without scope', ['Defer update to TODOS.md', 'Keep update']],
|
||||
])('ignores %s', (_name, labels) => expect(holdDeferKeepIndex(call(labels as string[]))).toBeNull());
|
||||
test('ignores multi-select and multi-question calls', () => {
|
||||
const multi = call(['Defer update to TODOS.md', 'Keep update in scope']);
|
||||
multi.questions[0].multiSelect = true;
|
||||
expect(holdDeferKeepIndex(multi)).toBeNull();
|
||||
const two = call(['Defer update to TODOS.md', 'Keep update in scope']);
|
||||
two.questions.push(structuredClone(two.questions[0]));
|
||||
expect(holdDeferKeepIndex(two)).toBeNull();
|
||||
expect(holdDeferKeepIndex(undefined)).toBeNull();
|
||||
});
|
||||
});
|
||||
@@ -976,3 +976,15 @@ export function nextCeoPostureContinuation(
|
||||
} else postureContinuations.set(seenQuestions, { modeId });
|
||||
return 'question';
|
||||
}
|
||||
|
||||
/** HOLD SCOPE's own "Deferring current scope" menu: one question, exactly a
|
||||
* Defer-to-TODOS option and a Keep-in-scope option. Returns the Keep index. */
|
||||
export function holdDeferKeepIndex(call: NativePlanQuestionCall | undefined): number | null {
|
||||
if (call?.questions.length !== 1) return null;
|
||||
const q = call.questions[0]!;
|
||||
if (q.multiSelect || q.options.length !== 2) return null;
|
||||
const labels = q.options.map(option => option.label.trim().replace(/^[A-Z][).:]\s+/, '').replace(/\s*\(recommended\)\s*$/i, ''));
|
||||
const defer = labels.findIndex(label => /^Defer\b[^\n]*\bTODOS(?:\.md)?$/i.test(label));
|
||||
const keep = labels.findIndex(label => /^Keep\b[^\n]*\bin scope$/i.test(label));
|
||||
return defer >= 0 && keep >= 0 && defer !== keep ? keep + 1 : null;
|
||||
}
|
||||
@@ -44,7 +44,7 @@ import {
|
||||
type AskUserQuestionFingerprint,
|
||||
type ClaudePtySession,
|
||||
} from './helpers/claude-pty-runner';
|
||||
import { ceoExpansionPacingChoice, ceoExpansionPacingReady, ceoModeSubmissionInput, hasNativePostAnswerCeoPosture, nextCeoModeNavigation, nextCeoPostureContinuation } from './helpers/ceo-mode-option';
|
||||
import { ceoExpansionPacingChoice, ceoExpansionPacingReady, ceoModeSubmissionInput, hasNativePostAnswerCeoPosture, holdDeferKeepIndex, nextCeoModeNavigation, nextCeoPostureContinuation } from './helpers/ceo-mode-option';
|
||||
import { createPlanCountFixture } from './helpers/plan-count-fixture';
|
||||
import { readPlanCountTranscript, type NativePublicToolEvent, type PlanCountTranscript } from './helpers/plan-count-transcript';
|
||||
import { readPendingQuestion, pendingQuestionRecorderStatus } from './helpers/plan-count-pending-question';
|
||||
@@ -286,10 +286,14 @@ describeE2E('/plan-ceo-review mode routing (gate)', () => {
|
||||
else {
|
||||
const pending = transcript.calls.find(call => !call.answered && !call.failed) ?? pendingQuestion;
|
||||
const question = capturePlanCountQuestion(currentInput, new Set(), 0, false, pending)!;
|
||||
if (c.mode === 'HOLD SCOPE' && question.nativeCall)
|
||||
// HOLD's own defer/keep menu (0G) is scope work, not the rigor decision
|
||||
// under assessment: keep the item in scope and assess the next decision.
|
||||
const keep = c.mode === 'HOLD SCOPE' ? holdDeferKeepIndex(question.nativeCall) : null;
|
||||
if (c.mode === 'HOLD SCOPE' && question.nativeCall && keep === null)
|
||||
continuedCallId ??= `${question.nativeCall.sessionId}:${question.nativeCall.toolUseId}`;
|
||||
const input = planCountQuestionInput(currentInput, question, 1);
|
||||
if (input.includes('\r')) await selectPtyNumberedOption(session, 1);
|
||||
const pick = keep ?? 1;
|
||||
const input = planCountQuestionInput(currentInput, question, pick);
|
||||
if (input.includes('\r')) await selectPtyNumberedOption(session, pick);
|
||||
else session.send(input);
|
||||
}
|
||||
continue;
|
||||
|
||||
Reference in new issue
Block a user