mirror of
https://github.com/KeygraphHQ/shannon.git
synced 2026-10-04 23:36:54 +02:00
* feat(worker): migrate agent runtime from Claude Agent SDK to pi harness * feat: remove Google Vertex AI provider support * fix(worker): route Bedrock and custom-base-URL providers from env * feat(prompts): instruct agents to call submit_exploitation_queue and submit_auth_result * fix(worker): count sub-agent cost and surface compaction failures * refactor(worker): rename claude-executor to pi-executor * feat(worker): pi-event-driven output formatting * fix(worker): gate adaptive thinking to Opus models, drop CLAUDE_THINKING_LEVEL * fix(worker): restore minLength/minItems on vuln-collector schemas * feat(worker): give task sub-agent write+bash, align tool descriptions * feat(worker): add glob custom tool and route code_path globs to it * refactor(prompts): use pi tool names (task, todo_write, read, bash, glob) * refactor(prompts): drop stale MCP terminology for collector tools * refactor(prompts): drop collector server names from deliverable instructions * fix(worker): restore minLength/minItems on pre-recon and exploit collector schemas * feat(worker): load playwright-cli skill via pi resource loader * refactor(cli): remove CLAUDE_CODE_MAX_OUTPUT_TOKENS config * build: drop @anthropic-ai/claude-code from worker image * docs: remove vertex references from llms context * docs(worker): update stale sdk comments * refactor(worker): unify provider precedence between preflight and executor * feat(worker): enforce bounded bash timeouts via pi extension * ci: bump the beta release line to 2.0.0 (#356) * fix(cli): pin npx command hints to beta tag * fix: render agent deliverables before the success commit so resume preserves them (#377) * feat(cli): restructure run folder and improve terminal UX (#383) * feat: surface report at run root and nest run internals under .shannon * feat: use plain-language wording in user-facing terminal messages * feat(cli): guide users to watch scan progress and surface report path on start * docs: sync run-folder layout and CLI wording across docs and comments * feat(cli): add version command reporting package version or git SHA * feat(cli): detect TTY for interactive prompts, color, and progress output * docs: document --yes flag, version command, and tty module * fix(cli): FORCE_COLOR precedence and plain uninstall --yes output * fix(cli): respect empty NO_COLOR * fix(cli): let NO_COLOR take precedence over FORCE_COLOR * docs: mark claude-code-router integration as removed * refactor(worker): converge shared core with shannon-oss (#388) * fix(worker): port keygraph shared-core correctness fixes * refactor(worker): adopt collectors/ and ai/pi/ layout; add task budget cap and cancellation * refactor(worker): drop inconsistent Collector "Server" suffix * refactor(worker): drop unused providerConfig/apiKey seams, resolve credentials from env only * refactor(worker): port oss code_path pattern expansion + external_directory allow * fix(worker): preserve dotfile paths in code_path avoid patterns (.env no longer stripped to env) * feat(worker): render Unprocessed Vulnerabilities section in exploit deliverable (align with oss) * feat(worker): request set_blind_spots for all vuln classes (align auth/ssrf with production prompts) * refactor(worker): adopt unified permissionSystem* naming and helper layout * refactor(worker): inline blind_spots into vuln deliverable section array * chore(worker): drop unused zod dependency (tree is typebox-native) * fix(worker): normalize base32 TOTP secret to accept padding and whitespace * refactor(worker): adopt shared toolResult helper and flatSchema naming in collectors * refactor(worker): use undefined over null in queue-schema builders * docs(worker): converge renderer/collector doc comments to current pi terminology * refactor(worker): adopt schema.ts cleanInput/stringEnum helpers in collectors * feat(worker): converge exploit-collector/renderer with vendored; capture and render overview for blocked findings * refactor(worker): converge session-tools/pipeline/exploitation-checker with vendored * refactor(worker): converge task-tool usage reporting with vendored onUsage callback * refactor(worker): converge structured output onto a submitTool executor channel * docs(worker): expand exploit-renderer docstring to match shannon-oss * docs(worker): adopt richer vuln-renderer docstring from shannon-oss * docs(worker): neutralize billing-detection wording for shannon-oss parity * fix(worker): verify checkpoint hash in the deliverables clone being reset * fix(worker): fail fast on malformed exploitation queue JSON * fix(worker): honor retryable flag when classifying exploitation-queue check failures * fix(worker): fail fast on corrupted session.json in run-scope validation * feat(worker): propagate Temporal cancellation signal into agent and auth pi sessions * fix(worker): mark exploit agent complete when exploitation is skipped so resume skips it * prompts: drop scan description from executive report prompt * refactor(worker): add createGenericSubmitTool for raw JSON-schema submit tools * refactor(worker): gate playwright-cli skill to browser agents via skillsOverride (adopt shannon-oss mechanism) * docs(worker): correct formatLogTime comment to UTC to match toISOString * refactor(worker): converge queue-schemas with shannon-oss (guarded count, decl order) * refactor(worker): converge task-tool with shannon-oss (byte-identical; modelRegistry optional) * fix(worker): use replaceLiteral for all prompt value insertions to prevent $-mangling * fix(worker): classify agent execution failures by error type instead of hardcoding validation * fix(worker): cap auth-failure detail at 250 chars to match shannon-oss * style(worker): apply biome formatting * refactor(worker): remove per-session task delegation cap from task tool * style(cli): collapse usage hint now that the beta tag is gone * chore: mark the pi harness migration as a breaking change BREAKING CHANGE: Google Vertex AI is no longer a supported provider. The CLAUDE_CODE_USE_VERTEX, ANTHROPIC_VERTEX_PROJECT, CLOUD_ML_REGION, and GOOGLE_APPLICATION_CREDENTIALS environment variables, along with the use_vertex, vertex_project, and cloud_ml_region config.toml keys, are removed. Vertex users must switch to Anthropic, AWS Bedrock, or a custom Anthropic-compatible base URL. The CLAUDE_CODE_MAX_OUTPUT_TOKENS environment variable and the max_output_tokens config.toml key are also removed.
288 lines
9.4 KiB
TypeScript
288 lines
9.4 KiB
TypeScript
// Copyright (C) 2025 Keygraph, Inc.
|
|
//
|
|
// This program is free software: you can redistribute it and/or modify
|
|
// it under the terms of the GNU Affero General Public License version 3
|
|
// as published by the Free Software Foundation.
|
|
|
|
/**
|
|
* Human-readable console formatting for the agent executor.
|
|
*
|
|
* Driven by the pi harness event stream: `turn_end` (assistant text) and
|
|
* `tool_execution_start` (structured tool calls). Unlike the previous harness —
|
|
* where tool calls were tool_use JSON embedded in assistant text and had to be
|
|
* parsed out — pi delivers tool name + args as discrete events, so formatting is
|
|
* a direct mapping.
|
|
*/
|
|
|
|
import { AGENTS } from '../session-manager.js';
|
|
import { extractAgentType, formatDuration } from '../utils/formatting.js';
|
|
import type { ExecutionContext } from './types.js';
|
|
|
|
interface ToolCallInput {
|
|
url?: string;
|
|
command?: string;
|
|
description?: string;
|
|
path?: string;
|
|
todos?: Array<{ status: string; content: string }>;
|
|
[key: string]: unknown;
|
|
}
|
|
|
|
/** Agent prefix used to attribute output when parallel agents interleave on one stream. */
|
|
export function getAgentPrefix(description: string): string {
|
|
const agentPrefixes: Record<string, string> = {
|
|
'injection-vuln': '[Injection]',
|
|
'xss-vuln': '[XSS]',
|
|
'auth-vuln': '[Auth]',
|
|
'authz-vuln': '[Authz]',
|
|
'ssrf-vuln': '[SSRF]',
|
|
'injection-exploit': '[Injection]',
|
|
'xss-exploit': '[XSS]',
|
|
'auth-exploit': '[Auth]',
|
|
'authz-exploit': '[Authz]',
|
|
'ssrf-exploit': '[SSRF]',
|
|
};
|
|
|
|
for (const [agentName, prefix] of Object.entries(agentPrefixes)) {
|
|
const agent = AGENTS[agentName as keyof typeof AGENTS];
|
|
if (agent && description.includes(agent.displayName)) {
|
|
return prefix;
|
|
}
|
|
}
|
|
|
|
if (description.includes('injection')) return '[Injection]';
|
|
if (description.includes('xss')) return '[XSS]';
|
|
if (description.includes('authz')) return '[Authz]'; // Check authz before auth
|
|
if (description.includes('auth')) return '[Auth]';
|
|
if (description.includes('ssrf')) return '[SSRF]';
|
|
|
|
return '[Agent]';
|
|
}
|
|
|
|
/** Extract domain from URL for display. */
|
|
function extractDomain(url: string): string {
|
|
try {
|
|
const urlObj = new URL(url);
|
|
return urlObj.hostname || url.slice(0, 30);
|
|
} catch {
|
|
return url.slice(0, 30);
|
|
}
|
|
}
|
|
|
|
/** Format a playwright-cli command (run via the bash tool) into a clean progress indicator. */
|
|
function formatBrowserAction(command: string): string | null {
|
|
const match = command.match(/playwright-cli\s+(?:-s=\S+\s+)?(\S+)(?:\s+(.*))?/);
|
|
if (!match) return null;
|
|
|
|
const subcommand = match[1];
|
|
const args = match[2] || '';
|
|
|
|
switch (subcommand) {
|
|
case 'open':
|
|
case 'goto': {
|
|
const domain = args.trim() ? extractDomain(args.trim()) : '';
|
|
return domain ? `🌐 Navigating to ${domain}` : '🌐 Opening browser';
|
|
}
|
|
case 'go-back':
|
|
return '⬅️ Going back';
|
|
case 'go-forward':
|
|
return '➡️ Going forward';
|
|
case 'reload':
|
|
return '🔄 Reloading page';
|
|
case 'click':
|
|
case 'dblclick':
|
|
return `🖱️ Clicking ${(args || 'element').slice(0, 25)}`;
|
|
case 'hover':
|
|
return `👆 Hovering over ${(args || 'element').slice(0, 20)}`;
|
|
case 'type':
|
|
return `⌨️ Typing ${(args || 'text').slice(0, 20)}`;
|
|
case 'press':
|
|
case 'keydown':
|
|
case 'keyup':
|
|
return `⌨️ Pressing ${args || 'key'}`;
|
|
case 'fill':
|
|
return `📝 Filling ${(args || 'field').slice(0, 25)}`;
|
|
case 'select':
|
|
return '📋 Selecting dropdown option';
|
|
case 'check':
|
|
case 'uncheck':
|
|
return `☑️ ${subcommand === 'check' ? 'Checking' : 'Unchecking'} ${(args || 'element').slice(0, 20)}`;
|
|
case 'upload':
|
|
return '📁 Uploading file';
|
|
case 'drag':
|
|
return '🖱️ Dragging element';
|
|
case 'snapshot':
|
|
return '📸 Taking page snapshot';
|
|
case 'screenshot':
|
|
return '📸 Taking screenshot';
|
|
case 'eval':
|
|
case 'run-code':
|
|
return '🔍 Running JavaScript analysis';
|
|
case 'console':
|
|
return '📜 Checking console logs';
|
|
case 'network':
|
|
return '🌐 Analyzing network traffic';
|
|
case 'tab-list':
|
|
case 'tab-new':
|
|
case 'tab-close':
|
|
case 'tab-select':
|
|
return `🗂️ ${subcommand.replace('tab-', '')} browser tab`;
|
|
case 'dialog-accept':
|
|
return '💬 Accepting dialog';
|
|
case 'dialog-dismiss':
|
|
return '💬 Dismissing dialog';
|
|
case 'pdf':
|
|
return '📄 Saving page as PDF';
|
|
case 'resize':
|
|
return `🖥️ Resizing browser ${args || ''}`.trim();
|
|
default:
|
|
return `🌐 Browser: ${subcommand}`;
|
|
}
|
|
}
|
|
|
|
/** Summarize a todo_write update into a clean progress indicator. */
|
|
function summarizeTodoUpdate(input: ToolCallInput | undefined): string | null {
|
|
if (!input?.todos || !Array.isArray(input.todos)) {
|
|
return null;
|
|
}
|
|
|
|
const todos = input.todos;
|
|
const recent = todos.filter((t) => t.status === 'completed').at(-1);
|
|
if (recent) {
|
|
return `✅ ${recent.content}`;
|
|
}
|
|
|
|
const current = todos.filter((t) => t.status === 'in_progress').at(0);
|
|
if (current) {
|
|
return `🔄 ${current.content}`;
|
|
}
|
|
|
|
return null;
|
|
}
|
|
|
|
export function detectExecutionContext(description: string): ExecutionContext {
|
|
const isParallelExecution = description.includes('vuln agent') || description.includes('exploit agent');
|
|
|
|
const useCleanOutput =
|
|
description.includes('Pre-recon agent') ||
|
|
description.includes('Recon agent') ||
|
|
description.includes('Executive Summary and Report Cleanup') ||
|
|
description.includes('vuln agent') ||
|
|
description.includes('exploit agent');
|
|
|
|
const agentType = extractAgentType(description);
|
|
const agentKey = description.toLowerCase().replace(/\s+/g, '-');
|
|
|
|
return { isParallelExecution, useCleanOutput, agentType, agentKey };
|
|
}
|
|
|
|
/** Format assistant turn text (from a pi `turn_end` event). */
|
|
export function formatAssistantOutput(
|
|
text: string,
|
|
context: ExecutionContext,
|
|
turnCount: number,
|
|
description: string,
|
|
): string[] {
|
|
if (!text.trim()) {
|
|
return [];
|
|
}
|
|
|
|
if (context.isParallelExecution) {
|
|
// Compact, attributed output for interleaved parallel agents.
|
|
return [`${getAgentPrefix(description)} ${text}`];
|
|
}
|
|
// Full turn output for sequential agents.
|
|
return [`\n Turn ${turnCount} (${description}):`, ` ${text}`];
|
|
}
|
|
|
|
/**
|
|
* Format a pi `tool_execution_start` event into a clean one-line progress indicator.
|
|
*
|
|
* Maps the common tool surfaces — `task` (sub-agent delegation), `todo_write`
|
|
* (plan updates), `bash` (incl. playwright-cli browser actions), read-only file
|
|
* tools, and the structured collector/submit tools — to friendly lines. Returns
|
|
* `[]` when there's nothing worth surfacing (e.g. a todo update with no active item).
|
|
*/
|
|
export function formatToolCall(
|
|
toolName: string,
|
|
args: Record<string, unknown> | undefined,
|
|
context: ExecutionContext,
|
|
description: string,
|
|
): string[] {
|
|
const input = (args ?? {}) as ToolCallInput;
|
|
let line: string | null;
|
|
|
|
if (toolName === 'task') {
|
|
line = `🚀 Launching ${input.description ?? 'sub-agent'}`;
|
|
} else if (toolName === 'todo_write') {
|
|
line = summarizeTodoUpdate(input);
|
|
} else if (toolName === 'bash') {
|
|
const command = typeof input.command === 'string' ? input.command : '';
|
|
line = command.includes('playwright-cli') ? formatBrowserAction(command) : `💻 ${command.slice(0, 60)}`;
|
|
} else if (toolName === 'read' || toolName === 'grep' || toolName === 'find' || toolName === 'ls') {
|
|
const path = typeof input.path === 'string' ? ` ${input.path.slice(0, 60)}` : '';
|
|
line = `📖 ${toolName}${path}`;
|
|
} else if (toolName.startsWith('set_') || toolName.startsWith('add_') || toolName.startsWith('submit_')) {
|
|
line = `📊 ${toolName.replace(/_/g, ' ')}`;
|
|
} else {
|
|
line = `🔧 ${toolName}`;
|
|
}
|
|
|
|
if (!line) return [];
|
|
|
|
if (context.isParallelExecution) {
|
|
return [`${getAgentPrefix(description)} ${line}`];
|
|
}
|
|
return [` ${line}`];
|
|
}
|
|
|
|
export function formatErrorOutput(
|
|
error: Error & { code?: string; status?: number },
|
|
context: ExecutionContext,
|
|
description: string,
|
|
duration: number,
|
|
sourceDir: string,
|
|
isRetryable: boolean,
|
|
): string[] {
|
|
const lines: string[] = [];
|
|
|
|
if (context.isParallelExecution) {
|
|
lines.push(`${getAgentPrefix(description)} Failed (${formatDuration(duration)})`);
|
|
} else if (context.useCleanOutput) {
|
|
lines.push(`${context.agentType} failed (${formatDuration(duration)})`);
|
|
} else {
|
|
lines.push(` pi agent failed: ${description} (${formatDuration(duration)})`);
|
|
}
|
|
|
|
lines.push(` Error Type: ${error.constructor.name}`);
|
|
lines.push(` Message: ${error.message}`);
|
|
lines.push(` Agent: ${description}`);
|
|
lines.push(` Working Directory: ${sourceDir}`);
|
|
lines.push(` Retryable: ${isRetryable ? 'Yes' : 'No'}`);
|
|
|
|
if (error.code) {
|
|
lines.push(` Error Code: ${error.code}`);
|
|
}
|
|
if (error.status) {
|
|
lines.push(` HTTP Status: ${error.status}`);
|
|
}
|
|
|
|
return lines;
|
|
}
|
|
|
|
export function formatCompletionMessage(
|
|
context: ExecutionContext,
|
|
description: string,
|
|
turnCount: number,
|
|
duration: number,
|
|
): string {
|
|
if (context.isParallelExecution) {
|
|
return `${getAgentPrefix(description)} Complete (${turnCount} turns, ${formatDuration(duration)})`;
|
|
}
|
|
|
|
if (context.useCleanOutput) {
|
|
return `${context.agentType.charAt(0).toUpperCase() + context.agentType.slice(1)} complete! (${turnCount} turns, ${formatDuration(duration)})`;
|
|
}
|
|
|
|
return ` pi agent completed: ${description} (${turnCount} turns) in ${formatDuration(duration)}`;
|
|
}
|