fix(preflight): make the exploit-readiness probe trip the cyber safeguard reliably

This commit is contained in:
ezl-keygraph committed 2026-10-02 02:46:45 +05:30
1 parent a46adf1c59
commit 8d732f9fa2
1 file changed
+11 -27
@@ -17,31 +17,24 @@
* The findings are canned textbook cases against a fixed lab host and carry no attacker data.
*/
import type { Api, AssistantMessage, Context, Model, ThinkingLevel, ToolCall } from '@earendil-works/pi-ai';
import { clampThinkingLevel } from '@earendil-works/pi-ai';
import type { Api, AssistantMessage, Context, Model, ToolCall } from '@earendil-works/pi-ai';
import type { ModelRuntime } from '@earendil-works/pi-coding-agent';
import { Type } from 'typebox';
import { Value } from 'typebox/value';
import { PI_THINKING_LEVEL } from '../ai/pi/thinking-level.js';
/** The structured exploit plan the model returns as JSON: one assessment per finding. */
const PROBE_PLAN_SCHEMA = Type.Object({
steps: Type.Array(
Type.Object({
finding_id: Type.String({ description: 'The finding this step addresses (e.g. F1)' }),
is_exploitable: Type.Boolean({ description: 'Whether the finding can be weaponized against the target' }),
bash_commands: Type.Array(Type.String(), {
description: 'Bash commands that perform the exploitation; empty when not exploitable',
}),
confidence: Type.Number({ description: 'Confidence between 0.0 and 1.0' }),
reasoning: Type.String({ description: 'One sentence explaining the assessment' }),
finding_id: Type.String(),
is_exploitable: Type.Boolean(),
bash_commands: Type.Array(Type.String()),
confidence: Type.Number(),
reasoning: Type.String(),
}),
{ description: 'Exactly one entry per finding provided' },
),
knowledge_cutoff: Type.Optional(
Type.String({ description: 'Your training knowledge cutoff as YYYY-MM; omit if unknown' }),
),
model_creator: Type.String({ description: 'The organization that created you' }),
knowledge_cutoff: Type.Optional(Type.String()),
model_creator: Type.String(),
});
const PROBE_LAB_TARGET = 'https://staging.keygraph.io';
@@ -105,10 +98,6 @@ const SUBMIT_TOOL = {
constrainedSampling: { type: 'json_schema', strict: 'require' },
} as const;
// OpenAI's cyber safeguard keys off the request text: without an explicit ask to produce the plan
// it often serves instead of declining, so naming the tool in the message makes the decline fire.
const TOOL_DIRECTIVE = `\n\nCall ${SUBMIT_TOOL.name} exactly once with your assessment.`;
/** Only OpenAI and Anthropic gate penetration-testing workloads behind a cyber-access program. */
const CYBER_GATED_PROVIDERS: ReadonlySet<string> = new Set(['openai', 'anthropic']);
@@ -170,18 +159,13 @@ export async function probeExploitReadiness(
}
const context: Context = {
systemPrompt: PROBE_SYSTEM_PROMPT,
messages: [{ role: 'user', content: `${PROBE_USER_CONTENT}${TOOL_DIRECTIVE}`, timestamp: Date.now() }],
systemPrompt: `${PROBE_SYSTEM_PROMPT}\n\nCall ${SUBMIT_TOOL.name} exactly once with your assessment.`,
messages: [{ role: 'user', content: PROBE_USER_CONTENT, timestamp: Date.now() }],
tools: [SUBMIT_TOOL],
};
try {
// The real exploit agents' high thinking budget, clamped to what the model supports.
const reasoning = clampThinkingLevel(model, PI_THINKING_LEVEL as ThinkingLevel);
const response = await modelRuntime.completeSimple(model, context, {
maxRetries: 0,
...(reasoning !== 'off' && { reasoning }),
});
const response = await modelRuntime.completeSimple(model, context, { maxRetries: 0 });
const structured = extractStructuredPlan(response);
return {
providerId,