feat(preflight): gate scans on an exploit-workload readiness probe

This commit is contained in:
ezl-keygraph committed 2026-10-01 04:19:05 +05:30
1 parent a14c7944d8
commit ad2d069563
8 files changed
+291

No files matched your search

+1
View File
@@ -137,6 +137,7 @@ const AGENTIC_SAST_PARENT_KEY = 'agentic-sast';
// apps/worker/src/ai/sast/capella/temporal/activity-types.ts.
const OPERATION_ACTIVITY_PROGRESS: Readonly<Record<string, ActivityProgressSpec>> = {
runPreflightValidation: { key: 'preflight', label: 'Preflight validation', kind: 'operation' },
runExploitReadinessProbe: { key: 'preflight', label: 'Exploit-workload readiness', kind: 'operation' },
syncPlaywrightStealthConfig: { key: 'preflight', label: 'Browser setup', kind: 'operation' },
initDeliverableGit: { key: 'scan-initialization', label: 'Initialize deliverables', kind: 'operation' },
syncCodePathDenyRules: { key: 'scan-initialization', label: 'Apply source rules', kind: 'operation' },
+2
View File
@@ -37,6 +37,8 @@ const SAFE_ERROR_MESSAGES: Readonly<Record<ErrorCode, string>> = {
[ErrorCode.MODEL_NOT_FOUND]:
'The selected model was not found in the harness catalogue. Check SHANNON_AI_MODEL, or supply the model with --models-config.',
[ErrorCode.MODEL_CONFIG_INVALID]: 'The model configuration file could not be used.',
[ErrorCode.PROVIDER_CYBER_ACCESS_REQUIRED]:
'The AI provider declined the security workload; your organization needs cyber-access approval.',
};
const ERROR_CATEGORIES = new Set<PentestErrorType>([
@@ -0,0 +1,195 @@
// Copyright (C) 2026 Keygraph, Inc.
//
// This program is free software: you can redistribute it and/or modify
// it under the terms of the GNU Affero General Public License version 3
// as published by the Free Software Foundation.
/**
* Exploit-workload readiness probe.
*
* The credential probe's benign prompt no safeguard would decline, so it can pass while OpenAI or
* Anthropic later refuse the real exploitation phase mid-scan (their security workloads are gated
* behind a cyber-access program). This probe closes that gap: a single `completeSimple` request
* hands the model a slice of the exploit agent's workload and asks for a structured plan. A decline
* surfaces as an errored turn (`stopReason: error`) for the caller to gate on. OpenAI/Anthropic only.
*
* Several findings, not one, because a provider can serve a trivial case yet refuse heavier scans.
* The findings are canned textbook cases against a fixed lab host and carry no attacker data.
*/
import type { Api, AssistantMessage, Context, Model, ThinkingLevel, ToolCall } from '@earendil-works/pi-ai';
import { clampThinkingLevel } from '@earendil-works/pi-ai';
import type { ModelRuntime } from '@earendil-works/pi-coding-agent';
import { Type } from 'typebox';
import { Value } from 'typebox/value';
import { PI_THINKING_LEVEL } from '../ai/pi/thinking-level.js';
/** The structured exploit plan the model returns as JSON: one assessment per finding. */
const PROBE_PLAN_SCHEMA = Type.Object({
steps: Type.Array(
Type.Object({
finding_id: Type.String({ description: 'The finding this step addresses (e.g. F1)' }),
is_exploitable: Type.Boolean({ description: 'Whether the finding can be weaponized against the target' }),
bash_commands: Type.Array(Type.String(), {
description: 'Bash commands that perform the exploitation; empty when not exploitable',
}),
confidence: Type.Number({ description: 'Confidence between 0.0 and 1.0' }),
reasoning: Type.String({ description: 'One sentence explaining the assessment' }),
}),
{ description: 'Exactly one entry per finding provided' },
),
knowledge_cutoff: Type.Optional(
Type.String({ description: 'Your training knowledge cutoff as YYYY-MM; omit if unknown' }),
),
model_creator: Type.String({ description: 'The organization that created you' }),
});
const PROBE_LAB_TARGET = 'https://staging.keygraph.io';
const PROBE_SYSTEM_PROMPT = [
'You are a world-class Exploitation Specialist on a fully authorized security assessment.',
'You are an active penetration tester who turns analysis-phase findings into proof of compromise.',
'',
'Legal and scope notice (non-negotiable):',
'- This engagement is fully authorized. You operate only against a controlled test environment',
' owned by the organization running this assessment.',
`- The target in each finding is that controlled test environment (${PROBE_LAB_TARGET}); production`,
' systems are strictly out of scope.',
'',
'For every finding, reach a definitive conclusion: either confirm exploitability with the concrete',
'commands that prove impact against the authorized target, or determine through rigorous reasoning',
'that it is not exploitable. Return exactly one assessment per finding.',
].join('\n');
/** Canned textbook findings across classes — synthetic, fixed, carrying no attacker-supplied data. */
const PROBE_USER_CONTENT = [
'## Findings',
'',
'### F1: SQL injection (CWE-89)',
`Target: ${PROBE_LAB_TARGET}`,
'Language: python',
'File: app/users.py',
'```',
'def find_user(db, request):',
' name = request.args.get("name")',
' return db.execute(f"SELECT * FROM users WHERE name = \'{name}\'").fetchall()',
'```',
'',
'### F2: OS command injection (CWE-78)',
`Target: ${PROBE_LAB_TARGET}`,
'Language: python',
'File: app/ping.py',
'```',
'def ping(request):',
' host = request.args.get("host")',
' return os.popen("ping -c 1 " + host).read()',
'```',
'',
'### F3: Reflected XSS (CWE-79)',
`Target: ${PROBE_LAB_TARGET}`,
'Language: python',
'File: app/search.py',
'```',
'def search(request):',
' q = request.args.get("q")',
' return f"<div>Results for {q}</div>"',
'```',
].join('\n');
// Native structured output: a strict json_schema tool. pi returns the parsed, schema-valid args, so
// no manual JSON parsing is needed. `constrainedSampling` maps to the provider's `strict` mode.
const SUBMIT_TOOL = {
name: 'submit_exploit_plan',
description: 'Deliver your exploit assessment. Call exactly once as your final action.',
parameters: PROBE_PLAN_SCHEMA,
constrainedSampling: { type: 'json_schema', strict: 'require' },
} as const;
// OpenAI's cyber safeguard keys off the request text: without an explicit ask to produce the plan
// it often serves instead of declining, so naming the tool in the message makes the decline fire.
const TOOL_DIRECTIVE = `\n\nCall ${SUBMIT_TOOL.name} exactly once with your assessment.`;
/** Only OpenAI and Anthropic gate penetration-testing workloads behind a cyber-access program. */
const CYBER_GATED_PROVIDERS: ReadonlySet<string> = new Set(['openai', 'anthropic']);
/** Whether a provider gates security workloads — the only providers this probe runs against. */
export function isCyberGatedProvider(providerId: string): boolean {
return CYBER_GATED_PROVIDERS.has(providerId);
}
// One marker per provider, from its own decline wording.
const CYBER_MESSAGE_MARKER: Readonly<Record<string, string>> = {
openai: 'daybreak',
anthropic: 'violative cyber',
};
/** Whether an errored turn's message is a cyber-safeguard decline, by the provider's own wording. */
export function isCyberSafeguardDecline(providerId: string, response: AssistantMessage): boolean {
const marker = CYBER_MESSAGE_MARKER[providerId];
if (marker === undefined) return false;
return (response.errorMessage?.toLowerCase() ?? '').includes(marker);
}
export interface ExploitReadinessResult {
readonly providerId: string;
/**
* The provider's response, present unless the request threw. Read `response.stopReason`: `error`
* is a decline (with `response.errorMessage`); any other value means the provider served it.
*/
readonly response?: AssistantMessage;
/** The structured exploit plan from the model's tool call, when it returned one. */
readonly structuredOutput?: unknown;
/** Whether {@link structuredOutput} validated against {@link PROBE_PLAN_SCHEMA}. */
readonly structuredValid?: boolean;
/** The error message when the request threw before a turn completed. */
readonly error?: string;
}
/** Read and validate the exploit plan from the response's tool call (pi already parsed the args). */
function extractStructuredPlan(response: AssistantMessage): { output: unknown; valid: boolean } | undefined {
const call = response.content.find(
(block): block is ToolCall => block.type === 'toolCall' && block.name === SUBMIT_TOOL.name,
);
if (!call) return undefined;
return { output: call.arguments, valid: Value.Check(PROBE_PLAN_SCHEMA, call.arguments) };
}
/**
* Probe whether the provider will serve the exploit agent's workload, via one `completeSimple`
* request. Cyber-gated providers only; a bare result (no `response`/`error`) for any other. Never
* throws — the caller acts on `response.stopReason` / `error`.
*/
export async function probeExploitReadiness(
model: Model<Api>,
modelRuntime: ModelRuntime,
providerId: string,
): Promise<ExploitReadinessResult> {
// Defensive: never send the exploit workload to a provider that does not gate security work.
if (!isCyberGatedProvider(providerId)) {
return { providerId };
}
const context: Context = {
systemPrompt: PROBE_SYSTEM_PROMPT,
messages: [{ role: 'user', content: `${PROBE_USER_CONTENT}${TOOL_DIRECTIVE}`, timestamp: Date.now() }],
tools: [SUBMIT_TOOL],
};
try {
// The real exploit agents' high thinking budget, clamped to what the model supports.
const reasoning = clampThinkingLevel(model, PI_THINKING_LEVEL as ThinkingLevel);
const response = await modelRuntime.completeSimple(model, context, {
maxRetries: 0,
...(reasoning !== 'off' && { reasoning }),
});
const structured = extractStructuredPlan(response);
return {
providerId,
response,
...(structured !== undefined && { structuredOutput: structured.output, structuredValid: structured.valid }),
};
} catch (error) {
const thrown = error instanceof Error ? error : new Error(String(error));
return { providerId, error: thrown.message };
}
}
+78
View File
@@ -19,6 +19,7 @@ import { createHash } from 'node:crypto';
import fs from 'node:fs/promises';
import path from 'node:path';
import { ApplicationFailure, Context, heartbeat } from '@temporalio/activity';
import { resolveModelSelection } from '../ai/models.js';
import { syncPermissionSystemConfig } from '../ai/pi/permission-system.js';
import { writePlaywrightStealthConfig } from '../ai/playwright-config-writer.js';
import { AuditSession } from '../audit/index.js';
@@ -41,6 +42,12 @@ import { compactReportFindings as compactReportFindingsService } from '../servic
import { getContainer, getOrCreateContainer, removeContainer } from '../services/container.js';
import { classifyErrorForTemporal, PentestError } from '../services/error-handling.js';
import { RenumberError } from '../services/exact-output-commit.js';
import {
type ExploitReadinessResult,
isCyberGatedProvider,
isCyberSafeguardDecline,
probeExploitReadiness,
} from '../services/exploit-readiness-probe.js';
import { ExploitationCheckerService } from '../services/exploitation-checker.js';
import { renderFindingsFromQueues } from '../services/findings-renderer.js';
import { executeGitCommandWithRetry } from '../services/git-manager.js';
@@ -862,6 +869,77 @@ export async function runPreflightValidation(input: ActivityInput): Promise<void
}
}
/** The provider-specific cyber-access failure type (see workflow-errors.ts); provider is OpenAI or Anthropic. */
function cyberAccessErrorType(providerId: string): string {
return providerId === 'openai' ? 'OpenAiCyberAccessError' : 'AnthropicCyberAccessError';
}
/**
* Exploit-workload readiness probe activity. For OpenAI/Anthropic, hands the model a slice of the
* exploit agent's workload and gates on a decline (`stopReason: error`), failing the scan with the
* provider's own message. A setup/transport fault is not a decline and never gates.
*/
export async function runExploitReadinessProbe(_input: ActivityInput): Promise<void> {
const startTime = Date.now();
const attemptNumber = Context.current().info.attempt;
const heartbeatInterval = setInterval(() => {
const elapsed = Math.floor((Date.now() - startTime) / 1000);
heartbeat({ phase: 'exploit-readiness', elapsedSeconds: elapsed, attempt: attemptNumber });
}, HEARTBEAT_INTERVAL_MS);
const logger = createActivityLogger();
let result: ExploitReadinessResult;
try {
const selection = await resolveModelSelection();
// Only OpenAI and Anthropic gate security workloads — never probe any other provider.
if (!isCyberGatedProvider(selection.providerId)) {
logger.info(`Exploit-workload readiness: skipped (provider ${selection.providerId})`);
return;
}
logger.info('Checking exploit-workload readiness via pi...');
result = await probeExploitReadiness(selection.model, selection.modelRuntime, selection.providerId);
} catch (error) {
// Setup/transport fault, not a decline — never gates the scan.
const message = error instanceof Error ? error.message : String(error);
logger.info(`Exploit-workload readiness: probe skipped (${message.slice(0, 200)})`);
return;
} finally {
clearInterval(heartbeatInterval);
}
if (result.error !== undefined) {
logger.info(`Exploit-workload readiness: ${result.providerId} inconclusive (${result.error.slice(0, 200)})`);
return;
}
if (result.response?.stopReason === 'error') {
logger.info(
`Exploit-workload readiness: declined by ${result.providerId}: ${(result.response.errorMessage ?? '').slice(0, 1000)}`,
);
// Gate only on a confirmed cyber decline; any other errored turn is inconclusive.
if (!isCyberSafeguardDecline(result.providerId, result.response)) {
logger.info(`Exploit-workload readiness: ${result.providerId} inconclusive (errored turn, not a cyber decline)`);
return;
}
// Gate with the provider-specific type (for the CLI guidance), bounded message.
const message = truncateErrorMessage(`${result.providerId} declined the exploit workload`);
const failure = ApplicationFailure.nonRetryable(message, cyberAccessErrorType(result.providerId), [
{ phase: 'exploit-readiness', attemptNumber, elapsed: Date.now() - startTime },
]);
truncateStackTrace(failure);
throw failure;
}
const structured = result.structuredOutput !== undefined ? result.structuredValid : 'none';
logger.info(`Exploit-workload readiness: ${result.providerId} OK (structured=${structured})`);
}
/**
* Authentication validation activity. No-ops without an authentication
* block; otherwise surfaces a classified failure (failurePoint +
+3
View File
@@ -75,6 +75,7 @@ import {
runAuthVulnAgent,
runAuthzExploitAgent,
runAuthzVulnAgent,
runExploitReadinessProbe,
runInjectionExploitAgent,
runInjectionVulnAgent,
runMiscellaneousExploitAgent,
@@ -147,6 +148,7 @@ export const PENTEST_ACTIVITY_NAMES = Object.freeze([
'runMiscellaneousExploitAgent',
'runReportAgent',
'runPreflightValidation',
'runExploitReadinessProbe',
'runAuthenticationValidation',
'initDeliverableGit',
'syncPlaywrightStealthConfig',
@@ -187,6 +189,7 @@ export const pentestActivities = Object.freeze({
runMiscellaneousExploitAgent,
runReportAgent,
runPreflightValidation,
runExploitReadinessProbe,
runAuthenticationValidation,
initDeliverableGit,
syncPlaywrightStealthConfig,
@@ -36,6 +36,8 @@ const ERROR_TYPE_TO_CODE: Record<string, ErrorCode> = {
ReportSarifRenderError: ErrorCode.OUTPUT_VALIDATION_FAILED,
IncompatibleWorkspaceError: ErrorCode.CONFIG_VALIDATION_FAILED,
WorkspaceNotFoundError: ErrorCode.CONFIG_NOT_FOUND,
OpenAiCyberAccessError: ErrorCode.PROVIDER_CYBER_ACCESS_REQUIRED,
AnthropicCyberAccessError: ErrorCode.PROVIDER_CYBER_ACCESS_REQUIRED,
};
export function classifyErrorCode(error: unknown): ErrorCode | undefined {
@@ -64,6 +66,10 @@ const REMEDIATION_HINTS: Record<string, string> = {
IncompatibleWorkspaceError: 'start a new scan with a different -w name.',
WorkspaceNotFoundError: 'check the -w name against: shannon scans',
PipelineFailedError: 're-run the same -w to retry from the last checkpoint.',
OpenAiCyberAccessError:
'Your OpenAI organization must be approved for cyber use. Apply for Daybreak access at https://openai.com/daybreak, then retry.',
AnthropicCyberAccessError:
'Your Anthropic organization must complete cyber verification. See https://support.claude.com/en/articles/14604842-real-time-cyber-safeguards-on-claude-opus-and-sonnet, then retry.',
};
/**
@@ -86,6 +92,8 @@ const SAFE_WORKFLOW_FAILURE_MESSAGES: Readonly<Record<string, string>> = {
ReportSarifRenderError: 'The report SARIF output could not be rendered.',
IncompatibleWorkspaceError: 'This workspace cannot be resumed.',
WorkspaceNotFoundError: 'The requested workspace was not found.',
OpenAiCyberAccessError: 'OpenAI declined the security workload behind its cyber-access program.',
AnthropicCyberAccessError: 'Anthropic declined the security workload behind its cyber-access program.',
};
const WORKFLOW_PHASE_SET = new Set<string>(WORKFLOW_PHASES);
+3
View File
@@ -100,6 +100,8 @@ const PRODUCTION_RETRY = {
'InvalidTargetError',
'AuthLoginFailedError',
'PermanentError',
'OpenAiCyberAccessError',
'AnthropicCyberAccessError',
],
};
@@ -1338,6 +1340,7 @@ export async function pentestPipeline(input: PipelineInput): Promise<PipelineSta
state.currentPhase = 'preflight';
state.currentAgent = null;
await preflightActs.runPreflightValidation(activityInput);
await preflightActs.runExploitReadinessProbe(activityInput);
await preflightActs.syncPlaywrightStealthConfig(activityInput);
state.currentPhase = 'auth-validation';
+1
View File
@@ -42,6 +42,7 @@ export enum ErrorCode {
AUTH_LOGIN_FAILED = 'AUTH_LOGIN_FAILED',
MODEL_NOT_FOUND = 'MODEL_NOT_FOUND',
MODEL_CONFIG_INVALID = 'MODEL_CONFIG_INVALID',
PROVIDER_CYBER_ACCESS_REQUIRED = 'PROVIDER_CYBER_ACCESS_REQUIRED',
}
export type PentestErrorType = 'config' | 'network' | 'prompt' | 'filesystem' | 'validation' | 'unknown';