mirror of
https://github.com/garrytan/gstack.git
synced 2026-10-03 01:46:55 +02:00
Merge remote-tracking branch 'origin/capy/rel-c' into capy/audit-fix-wave
This commit is contained in:
commit
2dae4bf944
25 files changed
+1158
-88
No files matched your search
@@ -149,7 +149,7 @@ function hasNativePostureProse(text: string, posture: RegExp): boolean {
|
||||
/** Finish the selected native mode packet before waiting for its answer. */
|
||||
export function ceoModeSubmissionInput(
|
||||
visible: string, selected: NativePlanQuestionCall | undefined, targetMode: CeoMode,
|
||||
transcript: PlanCountTranscript, submitted: Set<string>,
|
||||
transcript: PlanCountTranscript, submitted: Set<string>, screenText = '',
|
||||
): string | null {
|
||||
if (!selected || selected.answered || selected.failed || !selected.sessionId || !selected.toolUseId ||
|
||||
transcript.status !== 'ready' || selected.questions.length < 2 ||
|
||||
@@ -161,16 +161,28 @@ export function ceoModeSubmissionInput(
|
||||
const modeQuestions = selected.questions.filter(q => q.options.filter(o => modeTitle(o.label)).length >= 2);
|
||||
if (modeQuestions.length !== 1 || findCeoModeOption(modeQuestions[0]!.options.map((o, i) =>
|
||||
({index:i + 1, label:o.label})), targetMode) === null) return null;
|
||||
const bar = posturePacketBar(visible);
|
||||
if (!bar || !bar.answered.every(Boolean) || JSON.stringify(bar.headers) !== JSON.stringify(
|
||||
selected.questions.map(q => q.header.trim().replace(/\s+/g, ' '))) ||
|
||||
planCountSubmissionInput(visible) !== '\r') return null;
|
||||
const rawBar = [...visible.matchAll(/←[^\r\n]+✔\s*Submit\s*→/g)].at(-1)!;
|
||||
const preceding = visible.slice(0, rawBar.index);
|
||||
if (/```|~~~|^\s*>|\b(?:example|quoted|source)[^:\n]*:\s*$/im.test(preceding)) return null;
|
||||
const compact = (text: string) => text.replace(/\s+/g, '');
|
||||
const panel = compact(visible.slice(rawBar.index! + rawBar[0].length)
|
||||
.replace(/^[ \t]*[│┃] ?/gm, '').replace(/^[ \t]*[●⏺] ?/gm, ''));
|
||||
const quotedContext = /```|~~~|^\s*>|\b(?:example|quoted|source)[^:\n]*:\s*$/im;
|
||||
const bar = posturePacketBar(visible);
|
||||
let review: string;
|
||||
if (bar) {
|
||||
if (!bar.answered.every(Boolean) || JSON.stringify(bar.headers) !== JSON.stringify(
|
||||
selected.questions.map(q => q.header.trim().replace(/\s+/g, ' '))) ||
|
||||
planCountSubmissionInput(visible) !== '\r') return null;
|
||||
const rawBar = [...visible.matchAll(/←[^\r\n]+✔\s*Submit\s*→/g)].at(-1)!;
|
||||
if (quotedContext.test(visible.slice(0, rawBar.index))) return null;
|
||||
review = visible.slice(rawBar.index! + rawBar[0].length);
|
||||
} else {
|
||||
// A review taller than the terminal scrolls its tab bar and heading off
|
||||
// the viewport (run 36606688266). The viewport must still end at the
|
||||
// focused Submit prompt; the accumulated screen text then supplies the
|
||||
// one complete review panel, authenticated below exactly as with a bar.
|
||||
const heading = screenText.lastIndexOf('Review your answers');
|
||||
if (heading < 0 || !compact(visible).endsWith(BARLESS_SUBMIT_END) ||
|
||||
quotedContext.test(screenText.slice(0, heading).split('\n').slice(-3).join('\n'))) return null;
|
||||
review = screenText.slice(heading);
|
||||
}
|
||||
const panel = compact(review.replace(/^[ \t]*[│┃] ?/gm, '').replace(/^[ \t]*[●⏺] ?/gm, ''));
|
||||
// Authenticate the complete review panel against native questions and
|
||||
// offered answers. An intended keypress or a selected-mode echo is not an ACK.
|
||||
let prefixes = ['Reviewyouranswers'];
|
||||
|
||||
@@ -1987,7 +1987,7 @@ function conflictingDesignClosure(text: string): boolean {
|
||||
new RegExp(`(?:^|[.!?;]\\s+|\\n)(?:If|When|Once|Unless|Assuming|Provided)\\b[^.!?\\n]*\\b${owner}\\b`, 'i').test(text);
|
||||
}
|
||||
|
||||
function hasCompletePlanReport(expectedPlanPath: string, minimumMtime: number, maximumMtime: number,
|
||||
export function hasCompletePlanReport(expectedPlanPath: string, minimumMtime: number, maximumMtime: number,
|
||||
allowRunHeaderForFailure = false, requiredReview?: 'Design'): boolean {
|
||||
if (!path.isAbsolute(expectedPlanPath)) return false;
|
||||
try {
|
||||
|
||||
@@ -229,6 +229,18 @@ export function createEvalCollector(suite: string): EvalCollector | null {
|
||||
}
|
||||
|
||||
/** DRY helper to record an E2E test result into the eval collector. */
|
||||
/** Exit reasons for an API or transport failure (session-runner.ts). */
|
||||
const INFRA_EXIT_REASONS = new Set(['error_api', 'timeout_startup', 'error_output_stream']);
|
||||
|
||||
/** API/transport error or CLI crash before the first model turn: INFRA, never a
|
||||
* verdict on the product. Any assistant event or counted turn means the model
|
||||
* ran, so its refusal, timeout or wrong answer stays an ordinary failure. */
|
||||
export function isPreTurnInfraFailure(result: Pick<SkillTestResult, 'exitReason' | 'transcript' | 'costEstimate'>): boolean {
|
||||
return result.costEstimate.turnsUsed === 0
|
||||
&& (INFRA_EXIT_REASONS.has(result.exitReason) || /^exit_code_\d+$/.test(result.exitReason))
|
||||
&& !result.transcript.some(event => event?.type === 'assistant');
|
||||
}
|
||||
|
||||
export function recordE2E(
|
||||
evalCollector: EvalCollector | null,
|
||||
name: string,
|
||||
@@ -241,9 +253,11 @@ export function recordE2E(
|
||||
? `${result.toolCalls[result.toolCalls.length - 1].tool}(${JSON.stringify(result.toolCalls[result.toolCalls.length - 1].input).slice(0, 60)})`
|
||||
: undefined;
|
||||
|
||||
const passed = extra?.passed ?? (result.exitReason === 'success' && result.browseErrors.length === 0);
|
||||
evalCollector?.addTest({
|
||||
name, suite, tier: 'e2e',
|
||||
passed: result.exitReason === 'success' && result.browseErrors.length === 0,
|
||||
passed,
|
||||
...(!passed && isPreTurnInfraFailure(result) ? { failure_class: 'infra' as const } : {}),
|
||||
duration_ms: result.duration,
|
||||
cost_usd: result.costEstimate.estimatedCost,
|
||||
transcript: result.transcript,
|
||||
|
||||
@@ -117,6 +117,9 @@ export function isEngBatchingIssueAUQ(fp: AskUserQuestionFingerprint, priorCalls
|
||||
return !priorCalls.some(prior => batchingIssueNumber(prior) === issue);
|
||||
}
|
||||
|
||||
// The report's target declaration field (Target / Review target / Reviewed target, optionally qualified).
|
||||
const TARGET_FIELD = /^(?:Reviewed |Review )?target(?: \([^)\n]*\))?:/i;
|
||||
|
||||
/** A native brief can use its D number and topic while its stable R identity
|
||||
* lives in the required saved ledger. Count that owned choice, not a title
|
||||
* spelling. This does not approve the row or validate the implementation. */
|
||||
@@ -160,26 +163,31 @@ function recordedBatchingIssue(call: NativePlanQuestionCall, savedPlan: string):
|
||||
const rawSourceNames = [...(lines[1] ?? '').matchAll(/\b[\w./-]+\.md\b/g)];
|
||||
const directSource = sourceNames.length > 0 && sourceNames.every(name => name === 'PLAN.md') &&
|
||||
new Set([...metadata.matchAll(/\bPLAN\.md:([1-9]\d*(?:[-–][1-9]\d*)?)\b/g)].map(match => match[1])).size <= 1;
|
||||
const targetName = (s: string) => clean(s).replace(/^Eng(?:ineering)? review:\s*/i, '')
|
||||
const targetName = (s: string) => clean(s).replace(/^Eng(?:ineering)? review\s*[:—–-]\s*/i, '')
|
||||
.replace(/^Plan\s*[:—–-]\s*/i, '').toLowerCase();
|
||||
const named = [...(lines[1] ?? '').matchAll(/"(Plan:\s*[^"\n]+)"|“(Plan:\s*[^”\n]+)”/g)]
|
||||
.map(match => targetName(match[1] ?? match[2]!));
|
||||
const named = [...(lines[1] ?? '').matchAll(/"(Plan:\s*[^"\n]+)"|“(Plan:\s*[^”\n]+)”|\b[Pp]lan\s+"([^"\n]+)"|\b[Pp]lan\s+“([^”\n]+)”/g)]
|
||||
.map(match => targetName(match[1] ?? match[2] ?? match[3] ?? match[4]!));
|
||||
const titles = tokens.slice(0, start).filter(token => token.type === 'heading' && token.depth === 1);
|
||||
// Target declarations are fields, whatever their list or emphasis markup.
|
||||
const targetFields = tokens.slice(0, start).flatMap((token, at) => {
|
||||
if (token.type !== 'paragraph' || !currentHeading(at)) return [];
|
||||
if ((token.type !== 'paragraph' && token.type !== 'list') || !currentHeading(at)) return [];
|
||||
const previous = tokens.slice(0, at).filter(t => t.type !== 'space').at(-1);
|
||||
const quotedContext = /\b(?:quoted|copied|historical|example|hypothetical|archived)\b[^\n]*:\s*$/i;
|
||||
if (previous?.type === 'paragraph' && quotedContext.test(previous.raw)) return [];
|
||||
const parts = token.raw.split('\n');
|
||||
return parts.filter((line, i) => /^Reviewed target:/.test(line) &&
|
||||
const parts = token.raw.split('\n').map(line => line.replace(/^\s*(?:[-*+]|\d+[.)])\s+/, '').replace(/[*_]/g, '').trim());
|
||||
return parts.filter((line, i) => TARGET_FIELD.test(line) &&
|
||||
!parts.slice(0, i).some(part => quotedContext.test(part)));
|
||||
});
|
||||
const namedSource = !rawSourceNames.length && named.length === 1 && titles.length === 1 &&
|
||||
const targetFiles = targetFields.length === 1 ? [...targetFields[0]!.matchAll(/[\w./-]*[\w-]+\.md\b/g)].map(match => match[0]) : [];
|
||||
// An unsourced brief inherits the report's one current PLAN.md target; its
|
||||
// ledger record still supplies the cited finding. A brief that names its plan
|
||||
// must name the report title's plan, and an unfenced copy of that plan may
|
||||
// add its own H1 only when it names that same plan.
|
||||
const namedSource = !rawSourceNames.length && named.length <= 1 && titles.length >= 1 &&
|
||||
titles[0]!.type === 'heading' && currentHeading(tokens.indexOf(titles[0]!)) &&
|
||||
/^Eng(?:ineering)? review:\s*Plan\s*[:—–-]/i.test(clean(titles[0]!.text)) &&
|
||||
targetName(titles[0]!.text) === named[0] && targetFields.length === 1 &&
|
||||
/^Reviewed target:\s*`?PLAN\.md`?(?:\s|$)/.test(targetFields[0]!) &&
|
||||
[...targetFields[0]!.matchAll(/\b[\w./-]+\.md\b/g)].length === 1;
|
||||
targetFiles.length === 1 && targetFiles[0]!.split('/').at(-1) === 'PLAN.md' &&
|
||||
(named.length === 0 || /^Eng(?:ineering)? review\s*[:—–-]\s*\S/i.test(clean(titles[0]!.text)) &&
|
||||
titles.every(title => title.type === 'heading' && targetName(title.text) === named[0]));
|
||||
if (!directSource && !namedSource) return;
|
||||
const withdrawn = (value: string, owners: string) => new RegExp(
|
||||
`(?:^|[.!?;]\\s+|\\n)(?:Correction:\\s*)?(?:${owners}) (?:is|was|has been) ["“'‘]?(?:withdrawn|cancelled|canceled|rejected|superseded|resolved|closed|hypothetical|not current|no longer current)\\b`, 'i').test(prose(value, true));
|
||||
@@ -265,7 +273,7 @@ function recordedBatchingIssue(call: NativePlanQuestionCall, savedPlan: string):
|
||||
if (questions.length !== 1) continue;
|
||||
const inlineBrief = fields[questions[0]!]!.slice(marker.length).trim();
|
||||
const inline = Boolean(inlineBrief);
|
||||
if (!inline && (namedSource || clean(fields[questions[0]! + 1] ?? '') !== clean(title))) continue;
|
||||
if (!inline && clean(fields[questions[0]! + 1] ?? '') !== clean(title)) continue;
|
||||
const sources = [...finding[0]!.matchAll(/\b([\w./-]+\.md)(?::([1-9]\d*(?:[-–][1-9]\d*)?))?\b/g)];
|
||||
if (sources.length !== 1 || sources[0]![1] !== 'PLAN.md' ||
|
||||
!inline && !sources[0]![2] || source && sources[0]![2] !== source) continue;
|
||||
|
||||
@@ -512,9 +512,6 @@ export interface ArmJudgeScore {
|
||||
*/
|
||||
export const ARM_JUDGE_MODEL = CLAUDE_FRONTIER_EVAL_MODEL;
|
||||
|
||||
/** Bounded retry-on-malformed loop: total attempts, not extra retries. */
|
||||
export const ARM_JUDGE_ATTEMPTS = 2;
|
||||
|
||||
/**
|
||||
* Build the over-engineering rubric prompt. Exported (pure) so the free
|
||||
* selftest can verify prompt construction without any API call.
|
||||
@@ -587,10 +584,10 @@ export function parseArmJudgeResponse(raw: unknown): ArmJudgeScore {
|
||||
*
|
||||
* - Zero-diff arms are VALID scored cells: the agent built nothing, so the
|
||||
* score is deterministically 0/"none" — no API call.
|
||||
* - Bounded retry-on-malformed: ARM_JUDGE_ATTEMPTS total attempts. callJudge
|
||||
* already retries 429s internally; this loop covers malformed/refused JSON.
|
||||
* - One sample, never re-asked: a malformed or refused verdict is a failed
|
||||
* sample. callJudge's transport-level 429 backoff is not a verdict retry.
|
||||
* - `opts.call` is an injection seam so the free selftest can exercise the
|
||||
* retry bound without spending API money. Defaults to the real callJudge.
|
||||
* malformed path without spending API money. Defaults to the real callJudge.
|
||||
*/
|
||||
export async function armJudge(
|
||||
task: string,
|
||||
@@ -605,18 +602,10 @@ export async function armJudge(
|
||||
};
|
||||
}
|
||||
const call = opts?.call ?? callJudge;
|
||||
const prompt = buildArmJudgePrompt(task, diff);
|
||||
let lastError: unknown;
|
||||
for (let attempt = 1; attempt <= ARM_JUDGE_ATTEMPTS; attempt++) {
|
||||
try {
|
||||
const raw = await call<Record<string, unknown>>(prompt, ARM_JUDGE_MODEL);
|
||||
return parseArmJudgeResponse(raw);
|
||||
} catch (err) {
|
||||
lastError = err;
|
||||
}
|
||||
const raw = await call<Record<string, unknown>>(buildArmJudgePrompt(task, diff), ARM_JUDGE_MODEL);
|
||||
try {
|
||||
return parseArmJudgeResponse(raw);
|
||||
} catch (err) {
|
||||
throw new Error(`armJudge: malformed verdict (never resampled) — ${err instanceof Error ? err.message : String(err)}`);
|
||||
}
|
||||
throw new Error(
|
||||
`armJudge: no well-formed verdict after ${ARM_JUDGE_ATTEMPTS} attempts — `
|
||||
+ (lastError instanceof Error ? lastError.message : String(lastError)),
|
||||
);
|
||||
}
|
||||
@@ -284,7 +284,11 @@ function currentCreatePreview(preview: string, r: any, config: string, cwd: stri
|
||||
if(event.name!=='Write'||`${event.sessionId}:${event.toolUseId}`!==r.pendingId||event.input?.file_path!==r.expected||
|
||||
Date.parse(event.timestamp)<startedAt||typeof event.input.content!=='string'||
|
||||
Buffer.byteLength(event.input.content)>MAX_WRITE_INPUT_BYTES) return false;
|
||||
const source=event.input.content.split(/\r?\n/), rows=preview.split('\n');
|
||||
// A crop can keep the pane's file row and rule above the preview while its
|
||||
// "Create file" title scrolls away. That row must name the owned path.
|
||||
const header=/^ {0,3}(?![1-9]\d*(?:[ \t]|\n))(\S[^\n]*)\n[╌─━]{3,}[ \t]*\n/.exec(preview);
|
||||
if(header && path.resolve(cwd,header[1]!.trim())!==r.expected) return false;
|
||||
const source=event.input.content.split(/\r?\n/), rows=preview.slice(header?.[0].length ?? 0).split('\n');
|
||||
const numbered:Array<{line:number;text:string}>=[];
|
||||
let leading='';
|
||||
for(const row of rows) {
|
||||
|
||||
Reference in new issue
Block a user