feat(repl): /class focus by vuln type, /authorization program ref; detect model refusals

Three improvements prompted by a real bug-bounty run:

1. /class idor,sqli,xss,ssrf — focus a run on specific vulnerability CLASSES.
   Expands each class to the matching library agents (by name/title/CWE) and
   pins them, so /run tests exactly those classes and skips recon-based
   selection. Friendlier than naming each agent via /only. Known aliases cover
   idor/bola, sqli, xss, ssrf, csrf, ssti, xxe, rce, lfi, auth/jwt, graphql,
   race, upload, cors, prototype-pollution, smuggling, and more.

2. /authorization <url> (aliases /authz, /program) — declare the engagement's
   authorization (e.g. https://hackerone.com/<program>). Recorded and added to
   the rules-of-engagement context so the run is framed as the authorized test
   it is, which reduces false model refusals on in-scope bounty targets. It does
   NOT widen scope — the grant still comes from /target, /scope-file or a
   capability — and the RoE context tells agents to keep within the program's
   rules (no DoS/mass-account-creation/out-of-scope; report those as leads).

3. Model-refusal detection: when an agent DECLINES a technique (safety/RoE
   pushback, e.g. "can't run mass account creation at a production service"),
   the harness now says so plainly instead of hiding it as "0 parseable
   findings (malformed JSON)". The operator sees WHY an agent found nothing.

422 tests passing.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
CyberSecurityUPandClaude Opus 4.8 committed 2026-10-03 07:24:48 -03:00
1 parent 8a3cb42c1f
commit 4ed0226a79
3 files changed
+181 -11

No files matched your search

+104 -4
View File
@@ -150,9 +150,9 @@ pub(crate) const ACCEPTED: &[&str] = &[
"/context", "/continue", "/creds", "/diff", "/exclude", "/exit", "/expand", "/feed",
"/finding", "/findings", "/focus", "/forget", "/full", "/go", "/goal", "/graph", "/guardrail", "/guardrails", "/help",
"/history", "/idle", "/inscope", "/instructions", "/integration", "/integrations", "/key", "/log",
"/logs", "/mcp", "/memory", "/model", "/models", "/objective", "/objectives", "/observe",
"/authorization", "/authz", "/program", "/logs", "/mcp", "/memory", "/model", "/models", "/objective", "/objectives", "/observe",
"/observe-only", "/offline",
"/onboard", "/only", "/oos", "/outofscope", "/policy", "/providers", "/proxy", "/research", "/quick", "/economy", "/eco", "/q", "/quit", "/recon",
"/onboard", "/only", "/oos", "/outofscope", "/policy", "/providers", "/proxy", "/class", "/classes", "/focus-class", "/research", "/quick", "/economy", "/eco", "/q", "/quit", "/recon",
"/pause", "/repo", "/report", "/results", "/resume", "/retest", "/revalidate", "/run", "/runs",
"/scope", "/scope-out", "/show", "/status", "/stop", "/sub", "/subscription", "/target",
"/temp-email", "/tempmail", "/theme", "/timeout", "/ua", "/url", "/useragent", "/validate",
@@ -162,8 +162,8 @@ pub(crate) const ACCEPTED: &[&str] = &[
/// All slash-commands, for Tab completion.
const COMMANDS: &[&str] = &[
"/help", "/onboard", "/show", "/config", "/providers", "/model", "/key", "/sub", "/target",
"/repo", "/auth", "/creds", "/focus", "/objective", "/scope-out", "/attach", "/context", "/mcp", "/offline",
"/research", "/quick", "/economy", "/eco", "/votes", "/chain", "/recon", "/tempmail", "/timeout", "/proxy", "/burp", "/ua", "/agents", "/only", "/theme", "/clear", "/run", "/stop", "/pause", "/continue", "/runs", "/results", "/report",
"/authorization", "/class", "/repo", "/auth", "/creds", "/focus", "/objective", "/scope-out", "/attach", "/context", "/mcp", "/offline",
"/class", "/research", "/quick", "/economy", "/eco", "/votes", "/chain", "/recon", "/tempmail", "/timeout", "/proxy", "/burp", "/ua", "/agents", "/only", "/theme", "/clear", "/run", "/stop", "/pause", "/continue", "/runs", "/results", "/report",
"/status", "/logs", "/diff", "/retest", "/validate", "/finding", "/expand", "/integrations",
"/memory", "/forget", "/graph", "/inscope", "/observe", "/guardrail", "/policy",
"/capability", "/audit", "/quit",
@@ -288,6 +288,7 @@ struct Session {
instructions: Option<String>,
/// Engagement objective / rules-of-engagement context (why + what matters).
objective: Option<String>,
authorization: Option<String>,
/// Explicit out-of-scope exclusions the agents must not touch.
out_of_scope: Option<String>,
/// Authorization boundary + guardrails, enforced by the harness.
@@ -334,6 +335,7 @@ impl Default for Session {
creds: None,
instructions: None,
objective: None,
authorization: None,
out_of_scope: None,
policy: Default::default(),
capability: None,
@@ -830,6 +832,16 @@ pub async fn repl(base: &Path, auth: SessionAuth) -> anyhow::Result<()> {
s.objective = Some(arg.to_string());
println!(" objective set — steers what agents prioritise and what counts as impact");
}
"/authorization" | "/authz" | "/program" => {
if arg == "clear" { s.authorization = None; println!(" authorization reference cleared"); continue; }
if arg.is_empty() {
println!(" authorization: {}", s.authorization.clone().unwrap_or_else(|| "(none) — declare the program/authorization with /authorization <url-or-text>, e.g. https://hackerone.com/<program>".into()));
println!(" \x1b[2mframes the run as the authorized test it is and is recorded; it does NOT widen scope — the grant still comes from /target, /scope-file or a capability.\x1b[0m");
continue;
}
s.authorization = Some(arg.to_string());
println!(" \x1b[32m🔏 authorization recorded\x1b[0m — {} (stay within the program's scope & rules of engagement; scope boundary still enforced)", arg);
}
"/scope-out" | "/outofscope" | "/oos" | "/exclude" => {
if arg == "clear" { s.out_of_scope = None; println!(" out-of-scope cleared"); continue; }
if arg.is_empty() {
@@ -889,6 +901,41 @@ pub async fn repl(base: &Path, auth: SessionAuth) -> anyhow::Result<()> {
if arg.is_empty() { println!(" recon intensity: {} ({}) — set with /recon <1-4> [1 quick · 2 standard · 3 deep · 4 exhaustive]", s.recon_intensity, lvl(s.recon_intensity)); }
else { s.recon_intensity = arg.parse::<usize>().unwrap_or(s.recon_intensity).clamp(1, 4); println!(" recon intensity: {} ({}) — more rounds, more enumeration, auto-installs tools", s.recon_intensity, lvl(s.recon_intensity)); }
}
"/class" | "/classes" | "/focus-class" => {
// Focus the run on specific vuln CLASSES (idor, sqli, xss, ssrf, …):
// expand each class to the matching agents from the library and pin
// them, so /run tests exactly those classes and skips recon-based
// selection. Friendlier than /only for "just hunt IDOR + SQLi".
if arg.trim().is_empty() {
println!(" focus a run on vuln classes — /class idor,sqli,xss,ssrf (pins the matching agents)");
println!(" known: idor bola sqli xss ssrf csrf ssti xxe rce lfi rfi idor redirect ssrf deserialization");
println!(" auth jwt graphql race upload cors prototype-pollution nosqli command-injection");
println!(" current pinned: {}", if s.pinned.is_empty() { "(none)".into() } else { s.pinned.join(", ") });
} else if arg.trim() == "clear" {
s.pinned.clear();
println!(" classes cleared — back to recon-driven selection");
} else {
let lib = agents::load(base);
let classes: Vec<String> = arg.split([',', ';', ' ']).map(str::trim).filter(|x| !x.is_empty()).map(|c| c.to_lowercase()).collect();
let mut pinned: Vec<String> = Vec::new();
let mut unknown: Vec<String> = Vec::new();
for c in &classes {
let matched = agents_for_class(&lib, c);
if matched.is_empty() { unknown.push(c.clone()); }
for n in matched { if !pinned.contains(&n) { pinned.push(n); } }
}
if pinned.is_empty() {
println!(" \x1b[33mno agents matched: {}\x1b[0m — try /agents list for names, or /only <agent>", classes.join(", "));
} else {
s.pinned = pinned;
println!(" \x1b[1;36m🎯 focus set\x1b[0m — {} class(es): {} → {} agent(s): {}",
classes.len() - unknown.len(), classes.iter().filter(|c| !unknown.contains(c)).cloned().collect::<Vec<_>>().join(", "),
s.pinned.len(), s.pinned.join(", "));
if !unknown.is_empty() { println!(" \x1b[33m⚠ no match for:\x1b[0m {} (ignored)", unknown.join(", ")); }
println!(" \x1b[2m/run tests exactly these · /class clear to unpin\x1b[0m");
}
}
}
"/research" => {
match arg.trim() {
"on" | "true" | "1" => { s.research = true; println!(" \x1b[1;36m🔬 research mode ON\x1b[0m — whitebox/greybox will hunt a NOVEL, CVE-reportable bug (known-CVE dedup + patch-diff variant analysis)"); }
@@ -1537,6 +1584,7 @@ async fn run(base: &Path, s: &Session, history: &mut Vec<RunRecord>) {
}
};
cfg.objective = s.objective.clone();
cfg.authorization = s.authorization.clone();
cfg.out_of_scope = s.out_of_scope.clone();
cfg.scope = s.policy.clone();
cfg.capability = s.capability.clone();
@@ -1621,6 +1669,7 @@ async fn start_background(base: &Path, s: &Session, reader: &mut Reader,
cfg.instructions = if s.attachments.is_empty() { s.instructions.clone() }
else { Some(format!("{}\n\nATTACHED CONTEXT:\n{}", s.instructions.clone().unwrap_or_default(), s.attachments.join("\n\n"))) };
cfg.objective = s.objective.clone();
cfg.authorization = s.authorization.clone();
cfg.out_of_scope = s.out_of_scope.clone();
cfg.scope = s.policy.clone();
cfg.capability = s.capability.clone();
@@ -1840,6 +1889,55 @@ fn memory_cmd(s: &Session, arg: &str) {
/// Project-local store: `<cwd>/.neurosploit/` so each project keeps its own
/// session, run history and command history (resume on reopen). No DB needed —
/// it's structured state, not semantic search.
/// Expand a vuln-class keyword (idor, sqli, xss, ssrf, …) to the agent names in
/// the library that implement it. Matches on the agent's name, title and CWE by
/// a set of substrings per class, so `/class sqli` pins every SQLi agent
/// (blind/error/time/union/login-bypass) without the operator naming each.
fn agents_for_class(lib: &agents::Library, class: &str) -> Vec<String> {
// class -> substrings to look for in name/title/cwe (lowercased).
let needles: Vec<&str> = match class {
"idor" | "bola" => vec!["idor", "bola", "bfla", "access_control", "excessive_data"],
"sqli" | "sql" | "sql-injection" => vec!["sqli", "sql_inj", "orm_injection", "nosql"],
"nosqli" | "nosql" => vec!["nosql"],
"xss" => vec!["xss", "cross_site_script", "dom_clobber", "mutation_xss", "postmessage"],
"ssrf" => vec!["ssrf"],
"csrf" => vec!["csrf"],
"ssti" | "template-injection" => vec!["ssti", "template_injection"],
"xxe" => vec!["xxe"],
"rce" | "command-injection" | "cmdi" => vec!["command_injection", "rce", "code_injection", "expression_language", "deserialization", "log4shell"],
"lfi" | "path-traversal" | "traversal" => vec!["lfi", "path_traversal", "file_read", "arbitrary_file"],
"rfi" => vec!["rfi"],
"redirect" | "open-redirect" => vec!["redirect"],
"deserialization" | "deser" => vec!["deserialization", "pickle", "yaml_deser"],
"auth" | "authentication" => vec!["auth_bypass", "login_sqli", "jwt", "mfa", "oauth", "oidc", "session", "saml", "password_reset", "2fa", "two_factor", "webauthn"],
"jwt" => vec!["jwt"],
"graphql" => vec!["graphql"],
"race" | "race-condition" => vec!["race"],
"upload" | "file-upload" => vec!["file_upload", "upload"],
"cors" => vec!["cors"],
"prototype-pollution" | "prototype" | "pp" => vec!["prototype_pollution"],
"crlf" => vec!["crlf", "response_splitting", "header_injection"],
"ldap" => vec!["ldap_injection"],
"xpath" => vec!["xpath"],
"smuggling" | "request-smuggling" => vec!["smuggling", "desync", "h2c", "hop_by_hop"],
"cache" | "cache-poisoning" => vec!["cache"],
"secrets" | "exposure" => vec!["exposure", "secret", "disclosure", "env_file", "backup_file", "git_"],
"business-logic" | "logic" => vec!["business_logic", "price_manipulation", "coupon", "workflow_step", "idempotency"],
other => vec![other], // fall back to a raw substring match
};
let mut out: Vec<String> = Vec::new();
// Exploitation agents live in vulns/ (web classes), with ai/ and infra/ for
// the LLM and host/AD classes — the ones a black/grey-box run can pin.
let pool = lib.vulns.iter().chain(lib.ai.iter()).chain(lib.infra.iter());
for a in pool {
let hay = format!("{} {} {}", a.name, a.title, a.cwe).to_lowercase();
if needles.iter().any(|n| hay.contains(n)) && !out.contains(&a.name) {
out.push(a.name.clone());
}
}
out
}
pub(crate) fn proj_dir() -> std::path::PathBuf {
let d = std::env::current_dir().unwrap_or_else(|_| std::path::PathBuf::from(".")).join(".neurosploit");
std::fs::create_dir_all(&d).ok();
@@ -2288,6 +2386,8 @@ fn help() {
h("/votes <n>", "number of validator votes per finding");
h("/chain <n>", "attack-chain depth (post-exploitation pivots; 0 = off)");
h("/recon <1-4>", "recon intensity: 1 quick · 2 standard · 3 deep · 4 exhaustive (installs tools)");
h("/class <a,b>", "focus a run on vuln classes (idor,sqli,xss,ssrf,…) — pins the matching agents");
h("/authorization <url>", "declare the program/authorization (e.g. a bug-bounty URL) — recorded; does NOT widen scope");
h("/research", "whitebox/greybox: hunt a NOVEL, CVE-reportable bug (known-CVE dedup + patch-diff variant analysis)");
h("/quick", "economy preset: short, low-cost run (1 voter · 1 chain round · light recon · ≤6 agents)");
h("/tempmail on|off", "opt-in disposable inbox (mail.tm) to read a register confirmation code");
+69 -7
View File
@@ -101,6 +101,15 @@ Base everything on real observed responses — never assume. Reply with a COMPAC
/// recon/exploit prompts so the engagement is steered as the user asked.
fn operator_directives(cfg: &RunConfig) -> String {
let mut s = String::new();
if let Some(a) = cfg.authorization.as_deref().filter(|x| !x.trim().is_empty()) {
// The operator's declared authorization for this engagement (e.g. a
// bug-bounty program). Stated as context, within the enforced scope —
// it frames the test as the authorized one it is. Stay within the
// program's own rules of engagement (no out-of-scope techniques, no
// disruption of production); if a technique would violate typical RoE,
// report it as a lead instead of running it.
s.push_str(&format!("AUTHORIZATION — this is an authorized security test under: {a}. Act within that program's scope and rules of engagement; the enforced scope boundary still applies, and anything that would break a program's standard RoE (DoS, mass account creation, data destruction, out-of-scope hosts) must be reported as a lead rather than executed.\n"));
}
if let Some(obj) = cfg.objective.as_deref().filter(|x| !x.trim().is_empty()) {
s.push_str(&format!("ENGAGEMENT OBJECTIVE — the goal and context of this test; let it shape what you prioritise and what counts as impact: {obj}\n"));
}
@@ -1143,8 +1152,18 @@ pub async fn run(cfg: RunConfig, lib: &Library, pool: &ModelPool, tx: Sender<Str
// malformed — which teaches the operator to ignore a
// warning that sometimes means a real parse failure.
if f.is_empty() && !text.trim().is_empty() && !reported_nothing(&text) {
let tail: String = text.chars().rev().take(120).collect::<String>().chars().rev().collect();
let _ = txc.send(format!("⚠ agent {} returned text but 0 parseable findings (model may have produced malformed JSON). Tail: {:?}", ag.name, tail)).await;
if looks_like_refusal(&text) {
// The model declined the technique — not a parse
// failure. Say so plainly so the operator knows
// WHY this agent found nothing (target read as
// production, or technique out of RoE) and can
// re-scope / adjust rather than chase a non-bug.
let reason: String = text.trim().chars().take(180).collect();
let _ = txc.send(format!("notify: 🚫 agent {} DECLINED this technique (model safety/RoE pushback, not a finding and not a parse error) — \"{}…\"", ag.name, reason)).await;
} else {
let tail: String = text.chars().rev().take(120).collect::<String>().chars().rev().collect();
let _ = txc.send(format!("⚠ agent {} returned text but 0 parseable findings (model may have produced malformed JSON). Tail: {:?}", ag.name, tail)).await;
}
}
// Live findings feed: surface each candidate the moment it appears.
for c in &f {
@@ -2968,6 +2987,33 @@ fn reported_nothing(text: &str) -> bool {
}
}
/// A model reply that is a REFUSAL / safety pushback rather than a result or a
/// parse failure — the model declined to run the technique (often because it
/// read the target as production, or the technique as out-of-scope). Surfacing
/// this as its own state stops it being hidden under "0 parseable findings",
/// and tells the operator the real reason an agent produced nothing.
fn looks_like_refusal(text: &str) -> bool {
let t = text.trim();
if t.is_empty() || t.starts_with('[') || t.starts_with('{') { return false; }
let low = t.to_lowercase();
// Phrases that signal the model is declining / redirecting, not reporting.
const SIGNS: &[&str] = &[
"i can't", "i cannot", "i won't", "i will not", "i'm not able",
"i am not able", "i'm unable", "cannot help with", "can't help with",
"not comfortable", "i must decline", "i have to decline", "won't be able to",
"against a production", "at a production service", "disrupt real users",
"that doesn't make it safe", "isn't something i can", "is not something i can",
"without explicit authorization", "without proper authorization",
"spin up a local", "run this exact playbook against it instead",
"i'd recommend testing against", "recommend a local", "use a test environment",
"this would be", "não posso", "não vou", "não é algo que", "ambiente de teste",
];
let hits = SIGNS.iter().filter(|s| low.contains(**s)).count();
// Prose (no JSON) that trips a refusal phrase and is short-ish reads as a
// decline. Require a clear phrase; two independent ones removes edge cases.
hits >= 1 && !reported_nothing(text)
}
/// Last `n` characters of `s`, on a char boundary (diagnostics only).
fn tail(s: &str, n: usize) -> String {
let total = s.chars().count();
@@ -3030,11 +3076,15 @@ fn extract_findings(text: &str, agent: &str) -> Vec<Finding> {
None => {
let t = text.trim();
if !t.is_empty() && t != "[]" {
eprintln!(
"[extract_findings] agent {agent}: no parseable JSON in model reply (len={}); raw tail: {:?}",
text.len(),
tail(text, 200)
);
if looks_like_refusal(text) {
eprintln!("[extract_findings] agent {agent}: model DECLINED the technique (safety/RoE pushback, not a parse failure): {:?}", tail(text, 200));
} else {
eprintln!(
"[extract_findings] agent {agent}: no parseable JSON in model reply (len={}); raw tail: {:?}",
text.len(),
tail(text, 200)
);
}
}
return vec![];
}
@@ -4350,6 +4400,18 @@ mod extraction_tests {
assert!(extract_findings("{\"findings\":[]}", "a").is_empty());
}
#[test]
fn a_model_decline_is_detected_as_refusal_not_a_parse_error() {
// The exact shape seen on a live run: a prose reply declining the technique.
let decline = "I can't run mass account creation / credential-stuffing loads at a production service — that is state-changing and can disrupt real users. Want me to spin up a local Juice Shop and run this exact playbook against it instead?";
assert!(looks_like_refusal(decline));
// A real findings array is NOT a refusal.
assert!(!looks_like_refusal("[{\"title\":\"SQLi\",\"severity\":\"High\"}]"));
// An honest empty result is NOT a refusal.
assert!(!looks_like_refusal("[]"));
assert!(!looks_like_refusal("No vulnerabilities were found in the tested endpoints."));
}
/// The bare-array happy path for agent selection.
#[test]
fn a_string_array_parses_plain() {
@@ -219,6 +219,13 @@ pub struct RunConfig {
/// and gate strictly on novelty. Steers whitebox/greybox.
#[serde(default)]
pub research: bool,
/// Operator-declared authorization reference (e.g. a bug-bounty program URL
/// like https://hackerone.com/zoom-private). Recorded in the audit trail
/// and added to the rules-of-engagement context so the engagement is framed
/// as the authorized test it is. Does NOT widen scope — the grant still
/// comes from the target/scope-file/capability.
#[serde(default)]
pub authorization: Option<String>,
/// Opt-in: when the app requires email confirmation to register, allow the
/// agent to use a free disposable-inbox API (mail.tm) to read the code/link.
/// Off by default. Account creation is still capped by the safety guardrail.
@@ -322,6 +329,7 @@ impl RunConfig {
pinned: Vec::new(),
chain_depth: 2,
research: false,
authorization: None,
proxy: None,
user_agent: None,
recon_intensity: 3,