v3.4.1: CLI-only Rust harness — interactive wizard, smart selection, tool doctrine, Typst, status

- Remove Rust web server (axum/tower-http); CLI-only binary
- Verbose logging (-v) + unique run-id output folder runs/ns-<ts>-<target>/
- status.json lifecycle (running → complete) + ✓ COMPLETE summary
- Interactive wizard when run with no args; detailed --help with testphp/DVWA examples + Kali tip
- Tool-usage doctrine injected into recon/exploit prompts: curl + rustscan/nmap
  (apt/brew/cargo install guidance) + browser via Playwright when present, else curl
- Smart recon-aware selection: map recon signals → agent categories, only run
  matching agents; heuristic fallback when LLM selection is empty
- Cross-model false-positive validation: voting prefers a model other than the finder
- Playwright MCP auto-provision (npx) + per-backend support (claude/codex; gemini/grok degrade)
- Gemini provider (API + gemini CLI subscription)
- Typst report (report.typ + compiled report.pdf) via blank structured template
- Lenient finding parsing (confidence as word/number) — fixes empty-results bug
- bump version 3.4.0 -> 3.4.1

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
CyberSecurityUPandClaude Opus 4.8 committed 2026-06-24 19:34:13 -03:00
1 parent e565270f43
commit 96f00c1c68
15 files changed
+512 -969

No files matched your search

+3 -3
View File
@@ -1,4 +1,4 @@
//! NeuroSploit v3.4.0 harness — a robust multi-model runtime for the
//! NeuroSploit v3.4.1 harness — a robust multi-model runtime for the
//! markdown-driven autonomous pentest engine.
//!
//! The harness loads the `agents_md/` library, drives a *pool* of LLM models
@@ -16,8 +16,8 @@ pub mod types;
pub use agents::{Agent, Library};
pub use models::{
cli_binary_for, installed_cli_backends, provider_for, providers, write_mcp_config, ChatClient,
ModelRef, Provider,
cli_binary_for, ensure_playwright_mcp, installed_cli_backends, mcp_supported, provider_for,
providers, write_mcp_config, ChatClient, ModelRef, Provider,
};
pub use pipeline::{run_whitebox, RunOutput};
pub use pipeline::run;
+45 -9
View File
@@ -146,20 +146,23 @@ impl ChatClient {
let mut cmd = Command::new(bin);
match bin {
// Claude Code headless print mode (uses the Claude subscription login).
// Tool autonomy is always enabled so the agent can use its built-in
// tools (Bash/curl/etc.) to actually probe the target — Playwright MCP
// is an *optional* add-on, not a requirement.
"claude" => {
cmd.arg("-p").arg("--model").arg(model);
cmd.arg("-p").arg("--model").arg(model).arg("--dangerously-skip-permissions");
// Required to allow tool autonomy when running as root.
cmd.env("IS_SANDBOX", "1");
if let Some(mcp) = mcp_config {
cmd.arg("--mcp-config").arg(mcp).arg("--dangerously-skip-permissions");
// Required to allow tool autonomy when running as root.
cmd.env("IS_SANDBOX", "1");
cmd.arg("--mcp-config").arg(mcp);
}
}
// Codex non-interactive exec (uses the ChatGPT/Codex login), prompt on stdin.
"codex" => {
cmd.arg("exec").arg("--model").arg(model);
cmd.arg("exec").arg("--model").arg(model)
.arg("--dangerously-bypass-approvals-and-sandbox");
if let Some(mcp) = mcp_config {
cmd.arg("--config").arg(format!("mcp_config_file={mcp}"))
.arg("--dangerously-bypass-approvals-and-sandbox");
cmd.arg("--config").arg(format!("mcp_config_file={mcp}"));
}
cmd.arg("-");
}
@@ -173,13 +176,17 @@ impl ChatClient {
}
_ => {}
}
cmd.stdin(Stdio::piped()).stdout(Stdio::piped()).stderr(Stdio::piped());
cmd.stdin(Stdio::piped()).stdout(Stdio::piped()).stderr(Stdio::piped()).kill_on_drop(true);
let mut child = cmd.spawn().map_err(|e| anyhow!("spawn {} failed: {}", bin, e))?;
if let Some(mut stdin) = child.stdin.take() {
stdin.write_all(prompt.as_bytes()).await?;
// Drop closes stdin so the CLI processes the prompt and exits.
}
let out = child.wait_with_output().await?;
// Cap a single agentic CLI turn so a stuck tool-loop can't hang the run.
let out = match tokio::time::timeout(Duration::from_secs(600), child.wait_with_output()).await {
Ok(r) => r?,
Err(_) => return Err(anyhow!("{} subscription CLI timed out after 600s", bin)),
};
let stdout = String::from_utf8_lossy(&out.stdout).trim().to_string();
let stderr = String::from_utf8_lossy(&out.stderr);
if !out.status.success() {
@@ -229,6 +236,35 @@ pub fn installed_cli_backends() -> Vec<&'static str> {
["claude", "codex", "grok", "gemini"].into_iter().filter(|b| binary_in_path(b)).collect()
}
/// Does this provider's agentic CLI accept a Playwright MCP config?
/// Claude Code and Codex do; Gemini/Grok CLIs don't take an MCP-config flag, so
/// they fall back to their own built-in tools.
pub fn mcp_supported(provider: &str) -> bool {
matches!(provider, "anthropic" | "openai")
}
/// Best-effort ensure the Playwright MCP server is available locally. Requires
/// `npx`; pre-warms `@playwright/mcp` so the first agent call isn't a cold start.
/// Returns Err with a clear reason when it can't be provisioned (caller then
/// degrades to built-in tools).
pub fn ensure_playwright_mcp() -> Result<()> {
if !binary_in_path("npx") {
return Err(anyhow!("npx (Node.js) not found — install Node to use Playwright MCP"));
}
// `npx -y @playwright/mcp@latest --help` installs the package into the npx
// cache on first run; ignore non-zero exit (some versions lack --help) as long
// as the package resolves.
let out = std::process::Command::new("npx")
.args(["-y", "@playwright/mcp@latest", "--help"])
.stdout(Stdio::null())
.stderr(Stdio::null())
.status();
match out {
Ok(_) => Ok(()),
Err(e) => Err(anyhow!("could not provision @playwright/mcp via npx: {e}")),
}
}
/// Write a Playwright `.mcp.json` into `dir` and return its path, so the agentic
/// CLI can drive a real browser (DOM/JS/network/screenshots) during execution.
pub fn write_mcp_config(dir: &std::path::Path) -> std::io::Result<std::path::PathBuf> {
+134 -17
View File
@@ -11,6 +11,7 @@ use tokio::sync::mpsc::Sender;
/// Result of an engagement run.
#[derive(Default, Serialize)]
pub struct RunOutput {
pub target: String,
pub findings: Vec<Finding>,
pub agents_ran: Vec<String>,
pub candidates: usize,
@@ -19,7 +20,28 @@ pub struct RunOutput {
pub artifacts: Vec<String>,
}
const RECON_SYS: &str = "You are a web recon specialist. Map the target's attack surface and reply with a compact JSON object (tech, endpoints, auth, apis, ai_features). No prose.";
const RECON_SYS: &str = "You are a web recon specialist on an AUTHORIZED engagement. You have shell tools (curl etc.) — actively fetch the target, enumerate pages/params, and map the real attack surface. Do not ask for permission; proceed. Reply with a compact JSON object (tech, endpoints, params, auth, apis). No prose.";
/// Tool-usage doctrine prepended to recon/exploit prompts so the agent knows
/// exactly what it may use. Best run on Kali Linux (or the Kali Docker image),
/// where these tools are preinstalled.
fn tool_doctrine(mcp_on: bool) -> String {
let browser = if mcp_on {
"A Playwright MCP browser IS available — use it for JS-heavy pages, DOM/JS execution, and to PROVE client-side issues (e.g. XSS firing); capture screenshots as evidence."
} else {
"No browser MCP is available — use `curl` (and `wget`) for all HTTP interaction; render/inspect responses directly."
};
format!(
"TOOLING (authorized; best on Kali Linux or the kalilinux/kali-rolling Docker image):\n\
- HTTP: `curl` (headers, methods, params, cookies), `wget`.\n\
- Ports/services: `rustscan` if present, else `nmap`; if neither is installed you may \
install via apt (`apt install -y nmap`), brew, or cargo (`cargo install rustscan`) — \
otherwise probe common ports with `curl`/`nc`.\n\
- Content/params: `ffuf`, `gobuster`, `gau`, `katana` when available.\n\
- {browser}\n\
Use only what is installed; degrade gracefully. Never run destructive or DoS actions.\n\n"
)
}
const VOTE_SYS: &str = "You are an adversarial security validator. Decide if the candidate finding is a REAL, reproducible, exploitable vulnerability with proof. Reply with JSON {\"verdict\":\"confirmed\"|\"rejected\",\"reason\":\"...\"}. Default to rejected when uncertain.";
const CODE_VOTE_SYS: &str = "You are an adversarial source-code reviewer. Decide if the reported issue is a REAL vulnerability in the provided code (reachable, exploitable, not a false positive). Reply JSON {\"verdict\":\"confirmed\"|\"rejected\",\"reason\":\"...\"}.";
@@ -40,9 +62,14 @@ pub async fn run(cfg: RunConfig, lib: &Library, pool: &ModelPool, tx: Sender<Str
let _ = tx.send("recon: offline mode — skipping model calls".into()).await;
"{}".to_string()
} else {
match pool.complete(RECON_SYS, &format!("Target: {}", cfg.target)).await {
let recon_user = format!("{}Target: {}", tool_doctrine(pool.mcp_config.is_some()), cfg.target);
match pool.complete(RECON_SYS, &recon_user).await {
Ok((m, t)) => {
let _ = tx.send(format!("recon complete via {}", m.label())).await;
if cfg.verbose {
let snip: String = t.chars().take(280).collect();
let _ = tx.send(format!(" recon> {}", snip.replace('\n', " "))).await;
}
t
}
Err(e) => {
@@ -63,22 +90,24 @@ pub async fn run(cfg: RunConfig, lib: &Library, pool: &ModelPool, tx: Sender<Str
let _ = tx.send(format!("selected {} specialist agents (RL-ranked)", selected.len())).await;
let _ = tx.send("offline: no exploitation performed (provide API keys or --subscription to run live)".into()).await;
let artifacts = persist(&cfg, &recon, "", &[]);
return RunOutput { findings: vec![], agents_ran: selected.iter().map(|a| a.name.clone()).collect(), candidates: 0, recon, artifacts };
return RunOutput { target: cfg.target.clone(), findings: vec![], agents_ran: selected.iter().map(|a| a.name.clone()).collect(), candidates: 0, recon, artifacts };
}
// Use the model to pick the agents whose preconditions match the recon —
// the harness reasons about *which* specialists to run, not all of them.
let chosen = select_agents(pool, &recon, &ranked, &tx).await;
let selected: Vec<Agent> = {
let mut sel: Vec<Agent> = if chosen.is_empty() {
ranked.clone()
} else {
ranked.iter().filter(|a| chosen.iter().any(|c| c == &a.name)).cloned().collect()
};
let selected: Vec<Agent> = if !chosen.is_empty() {
let sel: Vec<Agent> =
ranked.iter().filter(|a| chosen.iter().any(|c| c == &a.name)).cloned().collect();
if sel.is_empty() {
sel = ranked.clone();
heuristic_select(&ranked, &recon, cap)
} else {
sel.into_iter().take(cap).collect()
}
sel.into_iter().take(cap).collect()
} else {
// LLM selection failed/empty → recon-keyword heuristic, not a blind flat list.
let _ = tx.send("selection empty — using recon-keyword heuristic".into()).await;
heuristic_select(&ranked, &recon, cap)
};
let _ = tx
.send(format!("intelligently selected {} agent(s) matching recon: {}", selected.len(),
@@ -87,6 +116,8 @@ pub async fn run(cfg: RunConfig, lib: &Library, pool: &ModelPool, tx: Sender<Str
// ---- 3. Exploit (parallel) -----------------------------------------
let target = cfg.target.clone();
let verbose = cfg.verbose;
let mcp_on = pool.mcp_config.is_some();
let raw: Vec<(String, String, Vec<Finding>)> = stream::iter(selected.iter().cloned())
.map(|ag| {
let target = target.clone();
@@ -94,10 +125,18 @@ pub async fn run(cfg: RunConfig, lib: &Library, pool: &ModelPool, tx: Sender<Str
let txc = tx.clone();
async move {
let user = format!(
"{}\n\nReply ONLY with a JSON array of confirmed findings (may be empty []). \
Each item: {{id,title,severity,cwe,endpoint,payload,evidence,impact,remediation,confidence}}.",
ag.user.replace("{target}", &target).replace("{recon_json}", &recon)
"AUTHORIZED engagement — you have explicit permission to test {target}. \
Do not ask for confirmation — proceed and PROVE each issue.\n\n\
{doctrine}{body}\n\nWhen done, reply with ONLY a JSON array of confirmed findings (may be empty []). \
Each item: {{id,title,severity,cwe,endpoint,payload,evidence,impact,remediation,confidence}}. \
`evidence` must contain the concrete proof (request/response excerpt).",
target = target,
doctrine = tool_doctrine(mcp_on),
body = ag.user.replace("{target}", &target).replace("{recon_json}", &recon),
);
if verbose {
let _ = txc.send(format!(" ▶ launching agent: {} ({})", ag.name, ag.title.replace(" Agent", ""))).await;
}
match pool.complete(&ag.system, &user).await {
Ok((m, text)) => {
let f = extract_findings(&text, &ag.name);
@@ -145,7 +184,7 @@ pub async fn run_whitebox(cfg: RunConfig, lib: &Library, pool: &ModelPool, tx: S
if cfg.offline || bytes == 0 {
let artifacts = persist(&cfg, "{}", &context, &[]);
return RunOutput { findings: vec![], agents_ran: selected.iter().map(|a| a.name.clone()).collect(), candidates: 0, recon: String::new(), artifacts };
return RunOutput { target: cfg.target.clone(), findings: vec![], agents_ran: selected.iter().map(|a| a.name.clone()).collect(), candidates: 0, recon: String::new(), artifacts };
}
let raw: Vec<(String, String, Vec<Finding>)> = stream::iter(selected.iter().cloned())
@@ -200,7 +239,12 @@ async fn select_agents(pool: &ModelPool, recon: &str, catalog: &[Agent], tx: &Se
match pool.complete(SELECT_SYS, &user).await {
Ok((m, text)) => {
let names = parse_string_array(&text);
let _ = tx.send(format!("agent selection via {} → {} agent(s) chosen", m.label(), names.len())).await;
if names.is_empty() {
let preview: String = text.chars().take(120).collect();
let _ = tx.send(format!("agent selection via {} returned no parseable list ({} chars): {}", m.label(), text.len(), preview.replace('\n', " "))).await;
} else {
let _ = tx.send(format!("agent selection via {} → {} agent(s) chosen", m.label(), names.len())).await;
}
names
}
Err(e) => {
@@ -217,16 +261,88 @@ fn parse_string_array(text: &str) -> Vec<String> {
}
}
/// Fallback agent selection when the LLM selector fails: score each agent by
/// keyword overlap between its name/title and the recon text, always seed a
/// black-box baseline of high-yield web classes, and take the top `cap`.
fn heuristic_select(ranked: &[Agent], recon: &str, cap: usize) -> Vec<Agent> {
const BASELINE: &[&str] = &[
"sqli_error", "sqli_blind", "sqli_union", "xss_reflected", "xss_stored", "xss_dom",
"command_injection", "lfi", "path_traversal", "ssrf", "idor", "open_redirect",
"auth_bypass", "csrf", "ssti", "file_upload", "xxe", "information_disclosure",
"security_headers", "cors_misconfig",
];
let r = recon.to_lowercase();
// Recon signal → agent-name substrings. Only agents whose surface the recon
// actually identified get the signal boost; the rest rely on the baseline.
let signals: &[(&str, &[&str])] = &[
("graphql", &["graphql"]),
("jwt", &["jwt"]),
("oauth", &["oauth", "oidc", "saml"]),
("\"jwt\"", &["jwt"]),
("api", &["api_", "bola", "bfla", "idor", "mass_assign", "rate_limit"]),
("upload", &["file_upload", "zip_slip"]),
("websocket", &["websocket"]),
("\"ws\"", &["websocket"]),
("graphql", &["graphql"]),
("aws", &["aws_", "s3_", "imds", "cloud_"]),
("gcp", &["gcp_", "gcs_", "metadata"]),
("azure", &["azure_"]),
("kubernetes", &["k8s_", "kubelet"]),
("docker", &["docker_", "container_"]),
("ai_features", &["llm_", "prompt_injection", "rag", "vector_db"]),
("chat", &["llm_", "prompt_injection"]),
("jinja", &["ssti"]),
("flask", &["ssti", "ssrf", "command_injection"]),
("php", &["lfi", "rfi", "sqli", "command_injection"]),
("template", &["ssti", "csti"]),
("redirect", &["open_redirect"]),
("login", &["auth_bypass", "brute_force", "sqli", "default_credentials"]),
("search", &["xss", "sqli"]),
("cache", &["cache", "smuggl"]),
];
let mut scored: Vec<(i32, &Agent)> = ranked
.iter()
.map(|a| {
let mut score = 0;
if BASELINE.contains(&a.name.as_str()) {
score += 4;
}
// recon-signal mapping: boost agents matching identified surface
for (sig, names) in signals {
if r.contains(sig) && names.iter().any(|n| a.name.contains(n)) {
score += 6;
}
}
// direct keyword overlap with recon text
for tok in a.name.split('_') {
if tok.len() >= 4 && r.contains(tok) {
score += 2;
}
}
(score, a)
})
.collect();
scored.sort_by(|x, y| y.0.cmp(&x.0));
let mut out: Vec<Agent> = scored.iter().filter(|(s, _)| *s > 0).map(|(_, a)| (*a).clone()).collect();
if out.is_empty() {
out = ranked.to_vec();
}
out.into_iter().take(cap).collect()
}
async fn validate(candidates: Vec<Finding>, pool: &ModelPool, sys: &str, vote_n: usize, tx: &Sender<String>) -> Vec<Finding> {
// Prefer a model other than the primary (likely finder) to adjudicate.
let finder = pool.candidates.first().map(|m| m.label());
let validated: Vec<Finding> = stream::iter(candidates.into_iter())
.map(|mut f| {
let txc = tx.clone();
let finder = finder.clone();
async move {
let q = format!(
"Finding: {} | severity {} | {} | at {} | payload {} | evidence {}",
f.title, f.severity, f.cwe, f.endpoint, f.payload, f.evidence
);
let (yes, total) = pool.vote(sys, &q, vote_n).await;
let (yes, total) = pool.vote(sys, &q, vote_n, finder.as_deref()).await;
f.validated = total > 0 && yes * 2 >= total;
f.votes = format!("{yes}/{total}");
if f.confidence == 0.0 && total > 0 {
@@ -268,6 +384,7 @@ async fn finish(cfg: RunConfig, _lib: &Library, recon: String, transcript: Strin
}
RunOutput {
target: cfg.target.clone(),
candidates: findings.len(),
findings,
agents_ran: selected.iter().map(|a| a.name.clone()).collect(),
+12 -2
View File
@@ -87,8 +87,18 @@ impl ModelPool {
/// Ask up to `n` distinct models the same yes/no validation question and
/// return (confirmations, total_votes). A model answering "yes"/"confirmed"
/// counts as a confirmation. Used to cut false positives.
pub async fn vote(&self, system: &str, user: &str, n: usize) -> (usize, usize) {
let panel: Vec<ModelRef> = self.candidates.iter().take(n.max(1)).cloned().collect();
///
/// `skip` names the model that produced the finding; when the panel has more
/// than one model, that model is moved to the back so a DIFFERENT model
/// adjudicates first (cross-model false-positive validation).
pub async fn vote(&self, system: &str, user: &str, n: usize, skip: Option<&str>) -> (usize, usize) {
let mut ordered: Vec<ModelRef> = self.candidates.clone();
if let Some(finder) = skip {
if ordered.len() > 1 {
ordered.sort_by_key(|m| m.label() == finder); // finder (true) sorts last
}
}
let panel: Vec<ModelRef> = ordered.into_iter().take(n.max(1)).collect();
let mut confirmed = 0usize;
let mut total = 0usize;
for m in &panel {
+67 -1
View File
@@ -1,4 +1,9 @@
use crate::types::Finding;
use std::path::{Path, PathBuf};
/// The blank, structured Typst template (rendering logic). Data (`meta`,
/// `findings`) is prepended by `typst_report` to make a self-contained file.
const TYPST_TEMPLATE: &str = include_str!("../../../templates/report.typ");
fn sev_rank(s: &str) -> u8 {
match s {
@@ -74,9 +79,70 @@ pub fn html(target: &str, findings: &[Finding]) -> String {
h4{{margin:12px 0 3px;font-size:12px;text-transform:uppercase;letter-spacing:.5px;color:#8b5cf6}}\
.b{{color:#8b5cf6;font-weight:800}}</style></head><body>\
<h1><span class=b>NeuroSploit</span> Penetration Test Report</h1>\
<div class=meta>Target: <b>{t}</b> · v3.4.0 Rust harness · multi-model validated</div>\
<div class=meta>Target: <b>{t}</b> · v3.4.1 Rust harness · multi-model validated</div>\
<div>{chips}</div><h2>Findings ({n})</h2>{body}\
<p class=meta>Authorized testing only. Findings confirmed by multi-model adversarial voting.</p></body></html>",
t = esc(target), chips = chips, n = sorted.len(), body = body,
)
}
// ===== Typst report =====
/// Is the `typst` binary available on PATH?
fn typst_available() -> bool {
std::env::var_os("PATH")
.map(|p| std::env::split_paths(&p).any(|d| d.join("typst").is_file()))
.unwrap_or(false)
}
fn sorted_findings(findings: &[Finding]) -> Vec<Finding> {
let mut v = findings.to_vec();
v.sort_by_key(|f| sev_rank(&f.severity));
v
}
/// Escape a string for embedding inside a Typst `"..."` literal (single line).
fn tq(s: &str) -> String {
let cleaned: String = s.replace('\\', "\\\\").replace('"', "\\\"").replace(['\n', '\r'], " ");
format!("\"{}\"", cleaned)
}
/// Generate a self-contained `report.typ` (data + bundled template) in `dir`
/// and compile it to `report.pdf` via the `typst` binary. Falls back to leaving
/// the `.typ` when `typst` is unavailable.
pub fn typst_report(target: &str, findings: &[Finding], dir: &Path) -> std::io::Result<PathBuf> {
std::fs::create_dir_all(dir)?;
let run_id = dir.file_name().and_then(|s| s.to_str()).unwrap_or("run").to_string();
let mut data = String::new();
data.push_str(&format!(
"#let meta = (target: {}, run_id: {}, generated: {}, model: {})\n",
tq(target), tq(&run_id), tq("NeuroSploit v3.4.1"), tq("multi-model")
));
data.push_str("#let findings = (\n");
for f in sorted_findings(findings) {
data.push_str(&format!(
" (severity: {}, title: {}, agent: {}, cwe: {}, cvss: {}, endpoint: {}, payload: {}, evidence: {}, impact: {}, remediation: {}, votes: {}, confidence: {}),\n",
tq(&f.severity), tq(&f.title), tq(&f.agent), tq(&f.cwe), tq(&f.cvss),
tq(&f.endpoint), tq(&f.payload), tq(&f.evidence), tq(&f.impact),
tq(&f.remediation), tq(&f.votes), f.confidence,
));
}
data.push_str(")\n\n");
let typ_path = dir.join("report.typ");
std::fs::write(&typ_path, format!("{data}{TYPST_TEMPLATE}"))?;
if typst_available() {
let pdf_path = dir.join("report.pdf");
match std::process::Command::new("typst")
.arg("compile").arg(&typ_path).arg(&pdf_path).output()
{
Ok(o) if o.status.success() && pdf_path.exists() => return Ok(pdf_path),
Ok(o) => eprintln!("typst compile failed: {}",
String::from_utf8_lossy(&o.stderr).lines().next().unwrap_or("").trim()),
Err(e) => eprintln!("typst not runnable: {e}"),
}
}
Ok(typ_path)
}
@@ -80,6 +80,9 @@ pub struct RunConfig {
/// Path to the RL reward state file.
#[serde(default)]
pub rl_path: Option<String>,
/// Verbose: log each agent as it launches, recon snippet, and votes.
#[serde(default)]
pub verbose: bool,
}
fn default_vote() -> usize {
@@ -101,6 +104,7 @@ impl RunConfig {
subscription: false,
workdir: None,
rl_path: None,
verbose: false,
}
}
}