mirror of
https://github.com/KeygraphHQ/shannon.git
synced 2026-10-03 23:06:51 +02:00
* feat(worker): record token, cache, and turn usage per agent * feat: replace model tiers with a single SHANNON_AI_MODEL across five providers * feat(cli): rebuild the setup wizard for provider and model selection * docs: document single-model selection and supported providers * feat(worker): use chat completions for OpenAI behind a custom base URL * feat: add SHANNON_AI_OPENAI_FORMAT to pick the wire API for OpenAI gateways * refactor(cli): drop endpoint path hints from the gateway format picker * feat(worker): enable pi in-session provider retry with retry-after backoff * refactor(worker): hand provider error classification to pi and drop the Anthropic ladders * refactor: remove the subscription retry preset and pipeline config section * fix(worker): validate Bedrock credentials with the same live probe as other providers * feat(worker): render the report from structured findings instead of agent-written markdown * fix(worker): dispose the credential probe session on every path * fix(worker): refuse to replace the assembled report with an empty one * refactor(worker): catch post-processing throws across the whole finalization block * revert(worker): drop the report zero-findings guard * docs(worker): correct the retry split and Bedrock credential claims * docs: regenerate llms-full.txt from current sources * feat(cli): build and run the npx flow from a clone * refactor(cli): flatten the setup summary output * feat(cli): reject runs with more than one provider configured * fix(worker): say a rejected bash call never ran * chore(cli): drop grok-4.3 and gpt-5.6-luna from the setup suggestions * feat(worker): capture structured finding locations for SARIF output * fix(worker): enumerate queue confidence so the report inherits it verbatim * feat(worker): give the reporting phase a mode-specific output schema * feat(worker): emit a SARIF 2.1.0 log for exploitative runs * fix(worker): correct SARIF locations and defer fingerprinting to the upload action * fix(worker): drop the confidence suffix from the analysis-mode summary list * feat(worker): give exploit findings a dedicated code location field * feat(worker): carry structured code locations from the vuln queue to the report * fix(worker): join code locations from the vuln queue instead of re-asking agents * fix(worker): spell out the finding_id to category mapping in the tool schema * feat: drop Google/Gemini as a supported AI provider * fix(worker): stop asking the report agent for code locations * docs: correct the provider list and drop the removed rate-limit settings * docs: add provider cyber safeguards and suggested models per provider * docs: document the SARIF output and the report rating thresholds
446 lines
19 KiB
TypeScript
446 lines
19 KiB
TypeScript
// Copyright (C) 2025 Keygraph, Inc.
|
|
//
|
|
// This program is free software: you can redistribute it and/or modify
|
|
// it under the terms of the GNU Affero General Public License version 3
|
|
// as published by the Free Software Foundation.
|
|
|
|
/**
|
|
* Exploit Collector tool factory (parameterized by vulnerability class and
|
|
* per-run valid-ID set).
|
|
*
|
|
* Exposes a single TypeBox-validated tool `add_exploit`, called once per
|
|
* processed vulnerability by the 5 exploit-* agents (injection, xss, auth,
|
|
* ssrf, authz). After the agent terminates, the host harvests
|
|
* collector.getAll() and runs exploit-renderer to produce
|
|
* {class}_exploitation_evidence.md. The collector state is the structured
|
|
* output.
|
|
*
|
|
* Schema shape:
|
|
* - The visible parameter schema is a single Type.Object with common fields
|
|
* required, status as a string union, and per-status fields marked optional
|
|
* at the tool layer (TypeBox cannot express a top-level discriminated union
|
|
* as the flat tool parameters). Each field's `description` text explains
|
|
* when it applies.
|
|
* - True per-status field enforcement runs inside the tool handler via a
|
|
* Type.Union([exploited, blocked]) re-validation using the TypeBox `Value`
|
|
* API. Missing-field errors come back to the agent as structured issues
|
|
* with retryable=true so it can fix and retry the call.
|
|
*
|
|
* Strict queue-ID validation: vulnerability_id is checked against the per-run
|
|
* queue's known IDs in the handler. Hallucinated or typo'd IDs are rejected
|
|
* with a structured error that includes the valid-ID list, letting the agent
|
|
* recover locally.
|
|
*
|
|
* Each field's description carries the bullet labels and reproducibility
|
|
* guidance, so the harness injects it into the agent's tool catalog.
|
|
*/
|
|
|
|
import { defineTool, type ToolDefinition } from '@earendil-works/pi-coding-agent';
|
|
import { type TSchema, Type } from 'typebox';
|
|
import { Value } from 'typebox/value';
|
|
import { stringEnum } from './schema.js';
|
|
|
|
// ============================================================================
|
|
// CLASS DISCRIMINATOR
|
|
// ============================================================================
|
|
|
|
export const EXPLOIT_VULN_CLASSES = ['injection', 'xss', 'auth', 'ssrf', 'authz'] as const;
|
|
export type VulnClass = (typeof EXPLOIT_VULN_CLASSES)[number];
|
|
|
|
// ============================================================================
|
|
// SCHEMA CONSTANTS
|
|
// ============================================================================
|
|
|
|
const SEVERITY_VALUES = ['critical', 'high', 'medium', 'low'] as const;
|
|
const CONFIDENCE_VALUES = ['high', 'medium', 'low'] as const;
|
|
|
|
const VALID_IDS_PREVIEW_LIMIT = 8;
|
|
|
|
function formatValidIdsPreview(validIds: ReadonlySet<string>): string {
|
|
const list = [...validIds];
|
|
const head = list.slice(0, VALID_IDS_PREVIEW_LIMIT).join(', ');
|
|
return list.length > VALID_IDS_PREVIEW_LIMIT ? `${head}, … (${list.length} total)` : head;
|
|
}
|
|
|
|
// ============================================================================
|
|
// PUBLIC TYPES (discriminated union — what consumers see)
|
|
// ============================================================================
|
|
|
|
export type ExploitedExploit = {
|
|
status: 'exploited';
|
|
vulnerability_id: string;
|
|
title: string;
|
|
vulnerable_location: string;
|
|
overview: string;
|
|
prerequisites?: string | null;
|
|
severity: (typeof SEVERITY_VALUES)[number];
|
|
impact: string;
|
|
exploitation_steps: string[];
|
|
proof_of_impact: string;
|
|
notes?: string | null;
|
|
};
|
|
|
|
export type BlockedExploit = {
|
|
status: 'blocked';
|
|
vulnerability_id: string;
|
|
title: string;
|
|
vulnerable_location: string;
|
|
overview: string;
|
|
prerequisites?: string | null;
|
|
confidence: (typeof CONFIDENCE_VALUES)[number];
|
|
current_blocker: string;
|
|
potential_impact: string;
|
|
evidence_of_vulnerability: string;
|
|
what_we_tried: string;
|
|
how_this_would_be_exploited: string[];
|
|
expected_impact: string;
|
|
notes?: string | null;
|
|
};
|
|
|
|
export type AddExploitInput = ExploitedExploit | BlockedExploit;
|
|
|
|
// ============================================================================
|
|
// SCHEMA BUILDER
|
|
// ============================================================================
|
|
|
|
export function buildSchemas(validIds: ReadonlySet<string>) {
|
|
const vulnerabilityIdField = Type.String({
|
|
minLength: 1,
|
|
description:
|
|
'Vulnerability identifier (e.g. "INJ-VULN-03"). Must match an ID from this run\'s ' +
|
|
'{class}_exploitation_queue.json exactly — the collector rejects IDs not in the queue. ' +
|
|
`Valid IDs for this run: ${formatValidIdsPreview(validIds)}.`,
|
|
});
|
|
|
|
const titleField = Type.String({
|
|
minLength: 1,
|
|
description:
|
|
'Descriptive vulnerability title (e.g. "SQL Injection — User Search", "IDOR — Unauthorized ' +
|
|
'Access to User Orders"). Concise; encodes the vulnerability category and where it lives.',
|
|
});
|
|
|
|
const vulnerableLocationField = Type.String({
|
|
minLength: 1,
|
|
description:
|
|
'Endpoint or mechanism where the vulnerability exists (e.g. "GET /api/products?id=", ' + '"POST /login").',
|
|
});
|
|
|
|
const overviewField = Type.String({
|
|
minLength: 1,
|
|
description:
|
|
'Brief summary of the exploit itself — what the vulnerability is and how it was demonstrated ' +
|
|
'(or how it would be demonstrated, for blocked findings). 1-3 sentences.',
|
|
});
|
|
|
|
const prerequisitesField = Type.Optional(
|
|
Type.Union([Type.String(), Type.Null()], {
|
|
description:
|
|
'Required setup, tools, or conditions to reproduce the exploit (e.g. authentication, ' +
|
|
'specific role, prior application state). Omit or pass null when no prerequisites apply.',
|
|
}),
|
|
);
|
|
|
|
const notesField = Type.Optional(
|
|
Type.Union([Type.String(), Type.Null()], {
|
|
description:
|
|
'Optional supplementary context — caveats, related findings, environmental observations. ' +
|
|
'Free-form Markdown. Omit or pass null when N/A.',
|
|
}),
|
|
);
|
|
|
|
const statusField = stringEnum(['exploited', 'blocked'], {
|
|
description:
|
|
'Verdict bucket. Set to "exploited" only after reaching Proof of Exploitation Level 3+ with ' +
|
|
'concrete impact evidence (extracted data, executed JavaScript, account takeover, internal ' +
|
|
'service access). Set to "blocked" only for real vulnerabilities where external factors ' +
|
|
'(NOT security defenses) prevented full exploitation. Findings where a security defense ' +
|
|
'successfully prevented exploitation after exhaustive bypass attempts are FALSE POSITIVE — ' +
|
|
'route those to your workspace tracking file, not this tool.',
|
|
});
|
|
|
|
// Per-status fields. All optional at the flat shape layer because a single
|
|
// Type.Object cannot express a top-level discriminated union; the handler
|
|
// re-validates against the strict union below for true enforcement.
|
|
const severityField = Type.Optional(
|
|
Type.Union([stringEnum(SEVERITY_VALUES), Type.Null()], {
|
|
description:
|
|
'REQUIRED when status="exploited". Severity of the demonstrated impact. Critical = Level 4 ' +
|
|
'(admin credentials extracted, sensitive data dumped, system commands executed, full account ' +
|
|
'takeover). High = Level 3 (data extraction proven, authentication bypass confirmed, ' +
|
|
'internal service access). Medium/Low based on impact narrowness or read-only access. Must ' +
|
|
'reflect demonstrated impact, not theoretical potential.',
|
|
}),
|
|
);
|
|
|
|
const impactField = Type.Optional(
|
|
Type.Union([Type.String({ minLength: 1 }), Type.Null()], {
|
|
description:
|
|
'REQUIRED when status="exploited". Business/security impact achieved by the exploit ' +
|
|
'(e.g. "Extracted full user table including bcrypt password hashes for 1,247 users", ' +
|
|
'"Achieved RCE as the application user; arbitrary shell commands executed"). Must describe ' +
|
|
'what was actually demonstrated, not what could theoretically happen.',
|
|
}),
|
|
);
|
|
|
|
const exploitationStepsField = Type.Optional(
|
|
Type.Union([Type.Array(Type.String({ minLength: 1 }), { minItems: 1 }), Type.Null()], {
|
|
description:
|
|
'REQUIRED when status="exploited". Ordered, reproducible exploitation steps — one Markdown ' +
|
|
'blob per numbered step. Each step must include full URLs (protocol + domain + port + path ' +
|
|
'+ params), complete payloads, and copy-paste-ready commands. Use clear placeholders for ' +
|
|
'variable values like [SESSION_TOKEN], [DATABASE_NAME], [TABLE_NAME], [TARGET_USER_ID]. ' +
|
|
'Write each step as natural Markdown — interleave prose with fenced code blocks (```bash, ' +
|
|
'```http, etc.) as you would in a write-up. Steps must be detailed enough that someone ' +
|
|
'unfamiliar with the application can follow without additional research.',
|
|
}),
|
|
);
|
|
|
|
const proofOfImpactField = Type.Optional(
|
|
Type.Union([Type.String({ minLength: 1 }), Type.Null()], {
|
|
description:
|
|
'REQUIRED when status="exploited". Concrete evidence of successful exploitation — extracted ' +
|
|
'data, achieved actions, captured request/response pairs, log excerpts. Markdown blob; ' +
|
|
'interleave prose with fenced code blocks. Must show what the exploit demonstrably achieved, ' +
|
|
'not theoretical impact.',
|
|
}),
|
|
);
|
|
|
|
const confidenceField = Type.Optional(
|
|
Type.Union([stringEnum(CONFIDENCE_VALUES), Type.Null()], {
|
|
description:
|
|
'REQUIRED when status="blocked". Confidence that this finding is a real vulnerability that ' +
|
|
'would be exploited if the external blocker were removed. High = code analysis strongly ' +
|
|
'confirms vulnerability and partial exploitation (Level 1-2) succeeded. Medium = code ' +
|
|
'analysis confirms but live evidence is partial. Low = signal-only; revisit if blocker is ' +
|
|
'removed in a future run.',
|
|
}),
|
|
);
|
|
|
|
const currentBlockerField = Type.Optional(
|
|
Type.Union([Type.String({ minLength: 1 }), Type.Null()], {
|
|
description:
|
|
'REQUIRED when status="blocked". What prevents full exploitation (e.g. "Server crashes after ' +
|
|
'5 requests, blocking enumeration", "OAuth callback requires verified third-party email ' +
|
|
'account we could not provision"). Must be an external operational constraint, not a ' +
|
|
'security defense.',
|
|
}),
|
|
);
|
|
|
|
const potentialImpactField = Type.Optional(
|
|
Type.Union([Type.String({ minLength: 1 }), Type.Null()], {
|
|
description:
|
|
'REQUIRED when status="blocked". What could be achieved if the blocker were removed (e.g. ' +
|
|
'"Full database read access", "Account takeover of arbitrary user via reset-token leak"). ' +
|
|
'Distinct from impact — this is the hypothetical outcome, not a demonstrated one.',
|
|
}),
|
|
);
|
|
|
|
const evidenceOfVulnerabilityField = Type.Optional(
|
|
Type.Union([Type.String({ minLength: 1 }), Type.Null()], {
|
|
description:
|
|
'REQUIRED when status="blocked". Code snippets, response excerpts, or observed behavior ' +
|
|
'proving the vulnerability is real. Markdown blob; interleave prose with fenced code blocks. ' +
|
|
'This is what convinces the reader the finding is not a false positive despite incomplete ' +
|
|
'exploitation.',
|
|
}),
|
|
);
|
|
|
|
const whatWeTriedField = Type.Optional(
|
|
Type.Union([Type.String({ minLength: 1 }), Type.Null()], {
|
|
description:
|
|
'REQUIRED when status="blocked". Log of attempted exploitation techniques and why each was ' +
|
|
'blocked. Each attempt should document the payload, the observed result, and the inferred ' +
|
|
'blocker. Markdown blob; multiple attempts as a list or distinct paragraphs. Demonstrates ' +
|
|
'exhaustive bypass effort per the Bypass Exhaustion Protocol.',
|
|
}),
|
|
);
|
|
|
|
const howThisWouldBeExploitedField = Type.Optional(
|
|
Type.Union([Type.Array(Type.String({ minLength: 1 }), { minItems: 1 }), Type.Null()], {
|
|
description:
|
|
'REQUIRED when status="blocked". Ordered hypothetical exploitation steps assuming the blocker ' +
|
|
'is removed — one Markdown blob per numbered step. Same reproducibility requirements as ' +
|
|
'exploitation_steps: full URLs, complete payloads, copy-paste-ready commands. Frame the ' +
|
|
'first step as "If [blocker] were removed: …".',
|
|
}),
|
|
);
|
|
|
|
const expectedImpactField = Type.Optional(
|
|
Type.Union([Type.String({ minLength: 1 }), Type.Null()], {
|
|
description:
|
|
'REQUIRED when status="blocked". Specific data or access that would be compromised if ' +
|
|
'exploitation succeeded (e.g. "Read access to all user profile data including PII; write ' +
|
|
'access to user-owned resources"). Markdown blob.',
|
|
}),
|
|
);
|
|
|
|
// The flat shape passed to defineTool. pi uses this to build the agent's
|
|
// tool catalog. Per-status enforcement happens in the handler via the
|
|
// strict union below.
|
|
const flatSchema = Type.Object({
|
|
status: statusField,
|
|
vulnerability_id: vulnerabilityIdField,
|
|
title: titleField,
|
|
vulnerable_location: vulnerableLocationField,
|
|
overview: overviewField,
|
|
prerequisites: prerequisitesField,
|
|
notes: notesField,
|
|
severity: severityField,
|
|
impact: impactField,
|
|
exploitation_steps: exploitationStepsField,
|
|
proof_of_impact: proofOfImpactField,
|
|
confidence: confidenceField,
|
|
current_blocker: currentBlockerField,
|
|
potential_impact: potentialImpactField,
|
|
evidence_of_vulnerability: evidenceOfVulnerabilityField,
|
|
what_we_tried: whatWeTriedField,
|
|
how_this_would_be_exploited: howThisWouldBeExploitedField,
|
|
expected_impact: expectedImpactField,
|
|
});
|
|
|
|
// Strict per-status validation. Re-runs in the handler so missing fields
|
|
// for the chosen status return a retryable error to the agent.
|
|
const ExploitedSchema = Type.Object({
|
|
status: Type.Literal('exploited'),
|
|
vulnerability_id: vulnerabilityIdField,
|
|
title: titleField,
|
|
vulnerable_location: vulnerableLocationField,
|
|
overview: overviewField,
|
|
prerequisites: prerequisitesField,
|
|
severity: stringEnum(SEVERITY_VALUES),
|
|
impact: Type.String({ minLength: 1 }),
|
|
exploitation_steps: Type.Array(Type.String({ minLength: 1 }), { minItems: 1 }),
|
|
proof_of_impact: Type.String({ minLength: 1 }),
|
|
notes: notesField,
|
|
});
|
|
|
|
const BlockedSchema = Type.Object({
|
|
status: Type.Literal('blocked'),
|
|
vulnerability_id: vulnerabilityIdField,
|
|
title: titleField,
|
|
vulnerable_location: vulnerableLocationField,
|
|
overview: overviewField,
|
|
prerequisites: prerequisitesField,
|
|
confidence: stringEnum(CONFIDENCE_VALUES),
|
|
current_blocker: Type.String({ minLength: 1 }),
|
|
potential_impact: Type.String({ minLength: 1 }),
|
|
evidence_of_vulnerability: Type.String({ minLength: 1 }),
|
|
what_we_tried: Type.String({ minLength: 1 }),
|
|
how_this_would_be_exploited: Type.Array(Type.String({ minLength: 1 }), { minItems: 1 }),
|
|
expected_impact: Type.String({ minLength: 1 }),
|
|
notes: notesField,
|
|
});
|
|
|
|
const StrictSchema = Type.Union([ExploitedSchema, BlockedSchema]);
|
|
|
|
return { flatSchema, StrictSchema };
|
|
}
|
|
|
|
// ============================================================================
|
|
// RESPONSE HELPERS
|
|
// ============================================================================
|
|
|
|
function toolResult(payload: Record<string, unknown>) {
|
|
return {
|
|
content: [{ type: 'text' as const, text: JSON.stringify(payload, null, 2) }],
|
|
details: undefined,
|
|
};
|
|
}
|
|
|
|
function successResult(data: Record<string, unknown>) {
|
|
return toolResult({ status: 'success', ...data });
|
|
}
|
|
|
|
function errorResult(message: string, errorType = 'ValidationError', retryable = true) {
|
|
return toolResult({ status: 'error', message, errorType, retryable });
|
|
}
|
|
|
|
function formatValueErrors(schema: TSchema, value: unknown): string {
|
|
const issues: string[] = [];
|
|
for (const err of Value.Errors(schema, value)) {
|
|
const path =
|
|
err.instancePath && err.instancePath.length > 0
|
|
? err.instancePath.replace(/^\//, '').replace(/\//g, '.')
|
|
: '(root)';
|
|
issues.push(`- ${path}: ${err.message}`);
|
|
}
|
|
return issues.join('\n');
|
|
}
|
|
|
|
// ============================================================================
|
|
// COLLECTOR FACTORY
|
|
// ============================================================================
|
|
|
|
export interface ExploitCollector {
|
|
tools: ToolDefinition[];
|
|
getAll(): AddExploitInput[];
|
|
}
|
|
|
|
export interface CreateExploitCollectorOptions {
|
|
vulnClass: VulnClass;
|
|
validIds: ReadonlySet<string>;
|
|
}
|
|
|
|
export function createExploitCollector(options: CreateExploitCollectorOptions): ExploitCollector {
|
|
const { vulnClass, validIds } = options;
|
|
const exploits: AddExploitInput[] = [];
|
|
const { flatSchema, StrictSchema } = buildSchemas(validIds);
|
|
|
|
const addExploitTool = defineTool({
|
|
name: 'add_exploit',
|
|
label: 'Add Exploit',
|
|
description:
|
|
`Record a single processed ${vulnClass} vulnerability as structured exploitation evidence. ` +
|
|
'Call this once per vulnerability in your queue.json after reaching a definitive verdict ' +
|
|
'(either successfully exploited or potential-but-blocked). The status field discriminates the ' +
|
|
"two report buckets; required sub-fields differ per status (see each field's description for " +
|
|
'which status requires it). Duplicate vulnerability_id calls are rejected — each vuln may only ' +
|
|
'be recorded once. Vulnerability IDs not in the queue.json are rejected with a list of valid ' +
|
|
'IDs. FALSE POSITIVE findings do NOT use this tool — they go to your workspace tracking file. ' +
|
|
'After all queue vulnerabilities have been emitted, the host renderer assembles the ' +
|
|
'deliverable Markdown from your recorded calls.',
|
|
parameters: flatSchema,
|
|
async execute(_toolCallId, input) {
|
|
// Re-validate against the strict discriminated union for per-status enforcement.
|
|
if (!Value.Check(StrictSchema, input)) {
|
|
return errorResult(
|
|
`Schema validation failed for status="${(input as { status?: string }).status}". ` +
|
|
'Required-field issues:\n' +
|
|
formatValueErrors(StrictSchema, input),
|
|
'ValidationError',
|
|
true,
|
|
);
|
|
}
|
|
const typed = Value.Clean(StrictSchema, structuredClone(input)) as AddExploitInput;
|
|
|
|
// Reject IDs not in this run's queue (typo'd or hallucinated).
|
|
if (!validIds.has(typed.vulnerability_id)) {
|
|
return errorResult(
|
|
`Vulnerability ID "${typed.vulnerability_id}" not in this run's queue. Valid IDs: ` +
|
|
`${formatValidIdsPreview(validIds)}. ` +
|
|
'Check the queue.json for the canonical ID — likely a typo or hallucinated ID.',
|
|
'ValidationError',
|
|
true,
|
|
);
|
|
}
|
|
|
|
const existing = exploits.find((e) => e.vulnerability_id === typed.vulnerability_id);
|
|
if (existing) {
|
|
return errorResult(
|
|
`Vulnerability ${typed.vulnerability_id} has already been recorded. Each vulnerability ` +
|
|
'may only be added once. Reach a final verdict before emitting.',
|
|
'DuplicateError',
|
|
false,
|
|
);
|
|
}
|
|
exploits.push(typed);
|
|
return successResult({ added: [typed.vulnerability_id], recorded_status: typed.status });
|
|
},
|
|
});
|
|
|
|
return {
|
|
tools: [addExploitTool],
|
|
getAll: (): AddExploitInput[] => [...exploits],
|
|
};
|
|
}
|