mirror of
https://github.com/KeygraphHQ/shannon.git
synced 2026-10-03 23:06:51 +02:00
* fix(worker): port keygraph shared-core correctness fixes * refactor(worker): adopt collectors/ and ai/pi/ layout; add task budget cap and cancellation * refactor(worker): drop inconsistent Collector "Server" suffix * refactor(worker): drop unused providerConfig/apiKey seams, resolve credentials from env only * refactor(worker): port oss code_path pattern expansion + external_directory allow * fix(worker): preserve dotfile paths in code_path avoid patterns (.env no longer stripped to env) * feat(worker): render Unprocessed Vulnerabilities section in exploit deliverable (align with oss) * feat(worker): request set_blind_spots for all vuln classes (align auth/ssrf with production prompts) * refactor(worker): adopt unified permissionSystem* naming and helper layout * refactor(worker): inline blind_spots into vuln deliverable section array * chore(worker): drop unused zod dependency (tree is typebox-native) * fix(worker): normalize base32 TOTP secret to accept padding and whitespace * refactor(worker): adopt shared toolResult helper and flatSchema naming in collectors * refactor(worker): use undefined over null in queue-schema builders * docs(worker): converge renderer/collector doc comments to current pi terminology * refactor(worker): adopt schema.ts cleanInput/stringEnum helpers in collectors * feat(worker): converge exploit-collector/renderer with vendored; capture and render overview for blocked findings * refactor(worker): converge session-tools/pipeline/exploitation-checker with vendored * refactor(worker): converge task-tool usage reporting with vendored onUsage callback * refactor(worker): converge structured output onto a submitTool executor channel * docs(worker): expand exploit-renderer docstring to match shannon-oss * docs(worker): adopt richer vuln-renderer docstring from shannon-oss * docs(worker): neutralize billing-detection wording for shannon-oss parity * fix(worker): verify checkpoint hash in the deliverables clone being reset * fix(worker): fail fast on malformed exploitation queue JSON * fix(worker): honor retryable flag when classifying exploitation-queue check failures * fix(worker): fail fast on corrupted session.json in run-scope validation * feat(worker): propagate Temporal cancellation signal into agent and auth pi sessions * fix(worker): mark exploit agent complete when exploitation is skipped so resume skips it * prompts: drop scan description from executive report prompt * refactor(worker): add createGenericSubmitTool for raw JSON-schema submit tools * refactor(worker): gate playwright-cli skill to browser agents via skillsOverride (adopt shannon-oss mechanism) * docs(worker): correct formatLogTime comment to UTC to match toISOString * refactor(worker): converge queue-schemas with shannon-oss (guarded count, decl order) * refactor(worker): converge task-tool with shannon-oss (byte-identical; modelRegistry optional) * fix(worker): use replaceLiteral for all prompt value insertions to prevent $-mangling * fix(worker): classify agent execution failures by error type instead of hardcoding validation * fix(worker): cap auth-failure detail at 250 chars to match shannon-oss * style(worker): apply biome formatting * refactor(worker): remove per-session task delegation cap from task tool
157 lines
6.3 KiB
TypeScript
157 lines
6.3 KiB
TypeScript
// Copyright (C) 2025 Keygraph, Inc.
|
|
//
|
|
// This program is free software: you can redistribute it and/or modify
|
|
// it under the terms of the GNU Affero General Public License version 3
|
|
// as published by the Free Software Foundation.
|
|
|
|
/**
|
|
* Generic `task` tool — pi.dev ships no built-in Task tool, so this supplies the
|
|
* Task-delegation surface Shannon's prompts require.
|
|
*
|
|
* Shannon's prompts mandate Task delegation (recon source tracer; the vuln
|
|
* agents delegate *every* code review; the exploit agents delegate automation),
|
|
* so this tool is required for parity, not optional. It spawns a nested pi
|
|
* session with the parent's resolved model object (never a tier string — that
|
|
* would route sub-agents through hardcoded IDs and leak billing), the parent's
|
|
* resource loader, and a fixed child tool surface.
|
|
*/
|
|
|
|
import type { ThinkingLevel } from '@earendil-works/pi-agent-core';
|
|
import { type AssistantMessage, type Model, Type } from '@earendil-works/pi-ai';
|
|
import {
|
|
type AuthStorage,
|
|
createAgentSession,
|
|
defineTool,
|
|
getAgentDir,
|
|
type ModelRegistry,
|
|
type ResourceLoader,
|
|
SessionManager,
|
|
SettingsManager,
|
|
type ToolDefinition,
|
|
} from '@earendil-works/pi-coding-agent';
|
|
|
|
export interface TaskToolContext {
|
|
cwd: string;
|
|
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
|
model: Model<any>;
|
|
thinkingLevel?: ThinkingLevel;
|
|
authStorage: AuthStorage;
|
|
/** Explicit model registry for sub-session resolution. Omit to inherit the parent's default. */
|
|
modelRegistry?: ModelRegistry;
|
|
resourceLoader: ResourceLoader;
|
|
cancellationSignal?: AbortSignal | undefined;
|
|
/**
|
|
* Reports the cost/tokens of each spawned sub-session back to the caller.
|
|
* Sub-agents run in their own pi sessions that the parent has no reference to,
|
|
* so without this their spend (the bulk of a whitebox run, since Shannon
|
|
* prompts delegate the heavy work) is invisible to billing.
|
|
*/
|
|
onUsage?: (usage: { cost: number; inputTokens: number; outputTokens: number }) => void;
|
|
}
|
|
|
|
const CHILD_TOOLS = ['read', 'grep', 'find', 'ls', 'write', 'bash'];
|
|
|
|
function textResult(text: string) {
|
|
return { content: [{ type: 'text' as const, text }], details: undefined };
|
|
}
|
|
|
|
export function createTaskTool(config: TaskToolContext): ToolDefinition {
|
|
const taskTool: ToolDefinition = defineTool({
|
|
name: 'task',
|
|
label: 'Task',
|
|
description:
|
|
'Delegate a focused task to a sub-agent that runs independently with its own tools and returns ' +
|
|
'the result. Use this to break complex work into smaller, parallelizable sub-tasks.',
|
|
executionMode: 'parallel',
|
|
promptSnippet: 'task - Delegate a focused task to a sub-agent with read, grep, find, ls, write, and bash.',
|
|
promptGuidelines: [
|
|
'Use the task tool to delegate focused work: code review, reconnaissance, automation scripting, validation.',
|
|
'Pass all necessary context in the "prompt" parameter — the sub-agent cannot see your conversation history.',
|
|
'The sub-agent can use read, grep, find, ls, write, and bash, but cannot call task or custom collector tools.',
|
|
'You can launch multiple task tool calls in a single message to run sub-tasks in parallel.',
|
|
],
|
|
parameters: Type.Object({
|
|
prompt: Type.String({
|
|
description: 'The task for the sub-agent to perform. Include all necessary context.',
|
|
}),
|
|
description: Type.Optional(Type.String({ description: 'A short (3-5 word) description of the task.' })),
|
|
}),
|
|
async execute(_toolCallId, params) {
|
|
const agentDir = getAgentDir();
|
|
const { session: subSession } = await createAgentSession({
|
|
cwd: config.cwd,
|
|
agentDir,
|
|
resourceLoader: config.resourceLoader,
|
|
model: config.model,
|
|
...(config.thinkingLevel && { thinkingLevel: config.thinkingLevel }),
|
|
tools: CHILD_TOOLS,
|
|
authStorage: config.authStorage,
|
|
...(config.modelRegistry && { modelRegistry: config.modelRegistry }),
|
|
sessionManager: SessionManager.inMemory(config.cwd),
|
|
settingsManager: SettingsManager.inMemory({
|
|
retry: { enabled: false },
|
|
compaction: { enabled: true },
|
|
}),
|
|
});
|
|
|
|
const abortChildSession = (): void => {
|
|
void subSession.abort().catch(() => {
|
|
// Parent logger is not available inside the tool; dispose still tears
|
|
// down the session if abort itself rejects.
|
|
});
|
|
};
|
|
const onCancellation = (): void => abortChildSession();
|
|
if (config.cancellationSignal?.aborted) {
|
|
abortChildSession();
|
|
} else {
|
|
config.cancellationSignal?.addEventListener('abort', onCancellation, { once: true });
|
|
}
|
|
|
|
let resultText = '';
|
|
let subCost = 0;
|
|
let subInputTokens = 0;
|
|
let subOutputTokens = 0;
|
|
subSession.subscribe((event) => {
|
|
if (event.type === 'turn_end') {
|
|
const msg = event.message as AssistantMessage | undefined;
|
|
for (const block of msg?.content ?? []) {
|
|
if (block.type === 'text' && block.text) {
|
|
resultText += (resultText ? '\n' : '') + block.text;
|
|
}
|
|
}
|
|
if (msg?.usage?.cost?.total != null) subCost += msg.usage.cost.total;
|
|
subInputTokens += msg?.usage?.input ?? 0;
|
|
subOutputTokens += msg?.usage?.output ?? 0;
|
|
}
|
|
});
|
|
|
|
let swallowedError: string | undefined;
|
|
try {
|
|
try {
|
|
await subSession.prompt(params.prompt);
|
|
} catch (err) {
|
|
const errorMsg = err instanceof Error ? err.message : String(err);
|
|
resultText += `\n[Sub-agent error: ${errorMsg}]`;
|
|
}
|
|
|
|
swallowedError = subSession.state.errorMessage;
|
|
// Read stats before dispose; reconcile cost the same way the parent does.
|
|
const subStats = subSession.getSessionStats();
|
|
if (subStats.cost > subCost) subCost = subStats.cost;
|
|
config.onUsage?.({ cost: subCost, inputTokens: subInputTokens, outputTokens: subOutputTokens });
|
|
} finally {
|
|
config.cancellationSignal?.removeEventListener('abort', onCancellation);
|
|
subSession.dispose();
|
|
}
|
|
|
|
if (swallowedError && !resultText.includes(swallowedError)) {
|
|
resultText += `\n[Sub-agent error: ${swallowedError}]`;
|
|
}
|
|
|
|
return textResult(resultText || '[Sub-agent produced no output]');
|
|
},
|
|
});
|
|
|
|
return taskTool;
|
|
}
|