Merge remote-tracking branch 'origin/capy/rel-c' into capy/audit-fix-wave

This commit is contained in:
garrytan committed 2026-09-29 20:23:51 +00:00
commit 2dae4bf944
25 files changed
+1158 -88

No files matched your search

+22 -10
View File
@@ -149,7 +149,7 @@ function hasNativePostureProse(text: string, posture: RegExp): boolean {
/** Finish the selected native mode packet before waiting for its answer. */
export function ceoModeSubmissionInput(
visible: string, selected: NativePlanQuestionCall | undefined, targetMode: CeoMode,
transcript: PlanCountTranscript, submitted: Set<string>,
transcript: PlanCountTranscript, submitted: Set<string>, screenText = '',
): string | null {
if (!selected || selected.answered || selected.failed || !selected.sessionId || !selected.toolUseId ||
transcript.status !== 'ready' || selected.questions.length < 2 ||
@@ -161,16 +161,28 @@ export function ceoModeSubmissionInput(
const modeQuestions = selected.questions.filter(q => q.options.filter(o => modeTitle(o.label)).length >= 2);
if (modeQuestions.length !== 1 || findCeoModeOption(modeQuestions[0]!.options.map((o, i) =>
({index:i + 1, label:o.label})), targetMode) === null) return null;
const bar = posturePacketBar(visible);
if (!bar || !bar.answered.every(Boolean) || JSON.stringify(bar.headers) !== JSON.stringify(
selected.questions.map(q => q.header.trim().replace(/\s+/g, ' '))) ||
planCountSubmissionInput(visible) !== '\r') return null;
const rawBar = [...visible.matchAll(/←[^\r\n]+✔\s*Submit\s*→/g)].at(-1)!;
const preceding = visible.slice(0, rawBar.index);
if (/```|~~~|^\s*>|\b(?:example|quoted|source)[^:\n]*:\s*$/im.test(preceding)) return null;
const compact = (text: string) => text.replace(/\s+/g, '');
const panel = compact(visible.slice(rawBar.index! + rawBar[0].length)
.replace(/^[ \t]*[│┃] ?/gm, '').replace(/^[ \t]*[●⏺] ?/gm, ''));
const quotedContext = /```|~~~|^\s*>|\b(?:example|quoted|source)[^:\n]*:\s*$/im;
const bar = posturePacketBar(visible);
let review: string;
if (bar) {
if (!bar.answered.every(Boolean) || JSON.stringify(bar.headers) !== JSON.stringify(
selected.questions.map(q => q.header.trim().replace(/\s+/g, ' '))) ||
planCountSubmissionInput(visible) !== '\r') return null;
const rawBar = [...visible.matchAll(/←[^\r\n]+✔\s*Submit\s*→/g)].at(-1)!;
if (quotedContext.test(visible.slice(0, rawBar.index))) return null;
review = visible.slice(rawBar.index! + rawBar[0].length);
} else {
// A review taller than the terminal scrolls its tab bar and heading off
// the viewport (run 36606688266). The viewport must still end at the
// focused Submit prompt; the accumulated screen text then supplies the
// one complete review panel, authenticated below exactly as with a bar.
const heading = screenText.lastIndexOf('Review your answers');
if (heading < 0 || !compact(visible).endsWith(BARLESS_SUBMIT_END) ||
quotedContext.test(screenText.slice(0, heading).split('\n').slice(-3).join('\n'))) return null;
review = screenText.slice(heading);
}
const panel = compact(review.replace(/^[ \t]*[│┃] ?/gm, '').replace(/^[ \t]*[●⏺] ?/gm, ''));
// Authenticate the complete review panel against native questions and
// offered answers. An intended keypress or a selected-mode echo is not an ACK.
let prefixes = ['Reviewyouranswers'];
+1 -1
View File
@@ -1987,7 +1987,7 @@ function conflictingDesignClosure(text: string): boolean {
new RegExp(`(?:^|[.!?;]\\s+|\\n)(?:If|When|Once|Unless|Assuming|Provided)\\b[^.!?\\n]*\\b${owner}\\b`, 'i').test(text);
}
function hasCompletePlanReport(expectedPlanPath: string, minimumMtime: number, maximumMtime: number,
export function hasCompletePlanReport(expectedPlanPath: string, minimumMtime: number, maximumMtime: number,
allowRunHeaderForFailure = false, requiredReview?: 'Design'): boolean {
if (!path.isAbsolute(expectedPlanPath)) return false;
try {
+15 -1
View File
@@ -229,6 +229,18 @@ export function createEvalCollector(suite: string): EvalCollector | null {
}
/** DRY helper to record an E2E test result into the eval collector. */
/** Exit reasons for an API or transport failure (session-runner.ts). */
const INFRA_EXIT_REASONS = new Set(['error_api', 'timeout_startup', 'error_output_stream']);
/** API/transport error or CLI crash before the first model turn: INFRA, never a
* verdict on the product. Any assistant event or counted turn means the model
* ran, so its refusal, timeout or wrong answer stays an ordinary failure. */
export function isPreTurnInfraFailure(result: Pick<SkillTestResult, 'exitReason' | 'transcript' | 'costEstimate'>): boolean {
return result.costEstimate.turnsUsed === 0
&& (INFRA_EXIT_REASONS.has(result.exitReason) || /^exit_code_\d+$/.test(result.exitReason))
&& !result.transcript.some(event => event?.type === 'assistant');
}
export function recordE2E(
evalCollector: EvalCollector | null,
name: string,
@@ -241,9 +253,11 @@ export function recordE2E(
? `${result.toolCalls[result.toolCalls.length - 1].tool}(${JSON.stringify(result.toolCalls[result.toolCalls.length - 1].input).slice(0, 60)})`
: undefined;
const passed = extra?.passed ?? (result.exitReason === 'success' && result.browseErrors.length === 0);
evalCollector?.addTest({
name, suite, tier: 'e2e',
passed: result.exitReason === 'success' && result.browseErrors.length === 0,
passed,
...(!passed && isPreTurnInfraFailure(result) ? { failure_class: 'infra' as const } : {}),
duration_ms: result.duration,
cost_usd: result.costEstimate.estimatedCost,
transcript: result.transcript,
+20 -12
View File
@@ -117,6 +117,9 @@ export function isEngBatchingIssueAUQ(fp: AskUserQuestionFingerprint, priorCalls
return !priorCalls.some(prior => batchingIssueNumber(prior) === issue);
}
// The report's target declaration field (Target / Review target / Reviewed target, optionally qualified).
const TARGET_FIELD = /^(?:Reviewed |Review )?target(?: \([^)\n]*\))?:/i;
/** A native brief can use its D number and topic while its stable R identity
* lives in the required saved ledger. Count that owned choice, not a title
* spelling. This does not approve the row or validate the implementation. */
@@ -160,26 +163,31 @@ function recordedBatchingIssue(call: NativePlanQuestionCall, savedPlan: string):
const rawSourceNames = [...(lines[1] ?? '').matchAll(/\b[\w./-]+\.md\b/g)];
const directSource = sourceNames.length > 0 && sourceNames.every(name => name === 'PLAN.md') &&
new Set([...metadata.matchAll(/\bPLAN\.md:([1-9]\d*(?:[-–][1-9]\d*)?)\b/g)].map(match => match[1])).size <= 1;
const targetName = (s: string) => clean(s).replace(/^Eng(?:ineering)? review:\s*/i, '')
const targetName = (s: string) => clean(s).replace(/^Eng(?:ineering)? review\s*[:—–-]\s*/i, '')
.replace(/^Plan\s*[:—–-]\s*/i, '').toLowerCase();
const named = [...(lines[1] ?? '').matchAll(/"(Plan:\s*[^"\n]+)"|“(Plan:\s*[^”\n]+)”/g)]
.map(match => targetName(match[1] ?? match[2]!));
const named = [...(lines[1] ?? '').matchAll(/"(Plan:\s*[^"\n]+)"|“(Plan:\s*[^”\n]+)”|\b[Pp]lan\s+"([^"\n]+)"|\b[Pp]lan\s+“([^”\n]+)”/g)]
.map(match => targetName(match[1] ?? match[2] ?? match[3] ?? match[4]!));
const titles = tokens.slice(0, start).filter(token => token.type === 'heading' && token.depth === 1);
// Target declarations are fields, whatever their list or emphasis markup.
const targetFields = tokens.slice(0, start).flatMap((token, at) => {
if (token.type !== 'paragraph' || !currentHeading(at)) return [];
if ((token.type !== 'paragraph' && token.type !== 'list') || !currentHeading(at)) return [];
const previous = tokens.slice(0, at).filter(t => t.type !== 'space').at(-1);
const quotedContext = /\b(?:quoted|copied|historical|example|hypothetical|archived)\b[^\n]*:\s*$/i;
if (previous?.type === 'paragraph' && quotedContext.test(previous.raw)) return [];
const parts = token.raw.split('\n');
return parts.filter((line, i) => /^Reviewed target:/.test(line) &&
const parts = token.raw.split('\n').map(line => line.replace(/^\s*(?:[-*+]|\d+[.)])\s+/, '').replace(/[*_]/g, '').trim());
return parts.filter((line, i) => TARGET_FIELD.test(line) &&
!parts.slice(0, i).some(part => quotedContext.test(part)));
});
const namedSource = !rawSourceNames.length && named.length === 1 && titles.length === 1 &&
const targetFiles = targetFields.length === 1 ? [...targetFields[0]!.matchAll(/[\w./-]*[\w-]+\.md\b/g)].map(match => match[0]) : [];
// An unsourced brief inherits the report's one current PLAN.md target; its
// ledger record still supplies the cited finding. A brief that names its plan
// must name the report title's plan, and an unfenced copy of that plan may
// add its own H1 only when it names that same plan.
const namedSource = !rawSourceNames.length && named.length <= 1 && titles.length >= 1 &&
titles[0]!.type === 'heading' && currentHeading(tokens.indexOf(titles[0]!)) &&
/^Eng(?:ineering)? review:\s*Plan\s*[:—–-]/i.test(clean(titles[0]!.text)) &&
targetName(titles[0]!.text) === named[0] && targetFields.length === 1 &&
/^Reviewed target:\s*`?PLAN\.md`?(?:\s|$)/.test(targetFields[0]!) &&
[...targetFields[0]!.matchAll(/\b[\w./-]+\.md\b/g)].length === 1;
targetFiles.length === 1 && targetFiles[0]!.split('/').at(-1) === 'PLAN.md' &&
(named.length === 0 || /^Eng(?:ineering)? review\s*[:—–-]\s*\S/i.test(clean(titles[0]!.text)) &&
titles.every(title => title.type === 'heading' && targetName(title.text) === named[0]));
if (!directSource && !namedSource) return;
const withdrawn = (value: string, owners: string) => new RegExp(
`(?:^|[.!?;]\\s+|\\n)(?:Correction:\\s*)?(?:${owners}) (?:is|was|has been) ["“'‘]?(?:withdrawn|cancelled|canceled|rejected|superseded|resolved|closed|hypothetical|not current|no longer current)\\b`, 'i').test(prose(value, true));
@@ -265,7 +273,7 @@ function recordedBatchingIssue(call: NativePlanQuestionCall, savedPlan: string):
if (questions.length !== 1) continue;
const inlineBrief = fields[questions[0]!]!.slice(marker.length).trim();
const inline = Boolean(inlineBrief);
if (!inline && (namedSource || clean(fields[questions[0]! + 1] ?? '') !== clean(title))) continue;
if (!inline && clean(fields[questions[0]! + 1] ?? '') !== clean(title)) continue;
const sources = [...finding[0]!.matchAll(/\b([\w./-]+\.md)(?::([1-9]\d*(?:[-–][1-9]\d*)?))?\b/g)];
if (sources.length !== 1 || sources[0]![1] !== 'PLAN.md' ||
!inline && !sources[0]![2] || source && sources[0]![2] !== source) continue;
+8 -19
View File
@@ -512,9 +512,6 @@ export interface ArmJudgeScore {
*/
export const ARM_JUDGE_MODEL = CLAUDE_FRONTIER_EVAL_MODEL;
/** Bounded retry-on-malformed loop: total attempts, not extra retries. */
export const ARM_JUDGE_ATTEMPTS = 2;
/**
* Build the over-engineering rubric prompt. Exported (pure) so the free
* selftest can verify prompt construction without any API call.
@@ -587,10 +584,10 @@ export function parseArmJudgeResponse(raw: unknown): ArmJudgeScore {
*
* - Zero-diff arms are VALID scored cells: the agent built nothing, so the
* score is deterministically 0/"none" — no API call.
* - Bounded retry-on-malformed: ARM_JUDGE_ATTEMPTS total attempts. callJudge
* already retries 429s internally; this loop covers malformed/refused JSON.
* - One sample, never re-asked: a malformed or refused verdict is a failed
* sample. callJudge's transport-level 429 backoff is not a verdict retry.
* - `opts.call` is an injection seam so the free selftest can exercise the
* retry bound without spending API money. Defaults to the real callJudge.
* malformed path without spending API money. Defaults to the real callJudge.
*/
export async function armJudge(
task: string,
@@ -605,18 +602,10 @@ export async function armJudge(
};
}
const call = opts?.call ?? callJudge;
const prompt = buildArmJudgePrompt(task, diff);
let lastError: unknown;
for (let attempt = 1; attempt <= ARM_JUDGE_ATTEMPTS; attempt++) {
try {
const raw = await call<Record<string, unknown>>(prompt, ARM_JUDGE_MODEL);
return parseArmJudgeResponse(raw);
} catch (err) {
lastError = err;
}
const raw = await call<Record<string, unknown>>(buildArmJudgePrompt(task, diff), ARM_JUDGE_MODEL);
try {
return parseArmJudgeResponse(raw);
} catch (err) {
throw new Error(`armJudge: malformed verdict (never resampled) — ${err instanceof Error ? err.message : String(err)}`);
}
throw new Error(
`armJudge: no well-formed verdict after ${ARM_JUDGE_ATTEMPTS} attempts — `
+ (lastError instanceof Error ? lastError.message : String(lastError)),
);
}
+5 -1
View File
@@ -284,7 +284,11 @@ function currentCreatePreview(preview: string, r: any, config: string, cwd: stri
if(event.name!=='Write'||`${event.sessionId}:${event.toolUseId}`!==r.pendingId||event.input?.file_path!==r.expected||
Date.parse(event.timestamp)<startedAt||typeof event.input.content!=='string'||
Buffer.byteLength(event.input.content)>MAX_WRITE_INPUT_BYTES) return false;
const source=event.input.content.split(/\r?\n/), rows=preview.split('\n');
// A crop can keep the pane's file row and rule above the preview while its
// "Create file" title scrolls away. That row must name the owned path.
const header=/^ {0,3}(?![1-9]\d*(?:[ \t]|\n))(\S[^\n]*)\n[╌─━]{3,}[ \t]*\n/.exec(preview);
if(header && path.resolve(cwd,header[1]!.trim())!==r.expected) return false;
const source=event.input.content.split(/\r?\n/), rows=preview.slice(header?.[0].length ?? 0).split('\n');
const numbered:Array<{line:number;text:string}>=[];
let leading='';
for(const row of rows) {