diff --git a/README.md b/README.md index a933eba..b67f74d 100755 --- a/README.md +++ b/README.md @@ -286,6 +286,59 @@ time. Two stores fix that, both under `.neurosploit/` in the project directory: harness derived itself are marked `inferred` and drawn dashed in the web console. Secrets never enter the graph; they stay in the vault. +### Scope: enforced, not requested + +`out_of_scope` used to be a sentence in the prompt and nothing checked it — a +*request* to the model, not a control. Scope is now a guard in code +(`crates/harness/src/scope.rs`): + +- **Hard scope** — an allowlist of hosts, `*.wildcards`, IPv4 CIDRs and URL + prefixes, plus exclusions that always win. It defaults to **the engagement's + target and nothing else**, so discovery can never widen the engagement: + finding a subdomain in a JS bundle is not authorization to test it. +- **Soft scope** — guardrails inside authorized territory: observe-only zones, + destructive HTTP verbs (off by default), account-creation cap, request-rate + guard, and payload classes that are never acceptable (data destruction, DoS) + — refused even against an in-scope host. +- Findings proven against a host outside the boundary are **withheld from the + report** and written to `out-of-scope-findings.json` as an incident to + disclose. + +``` +/inscope *.example.com 10.0.0.0/24 # authorize more +/scope-out payments.example.com # host-shaped entries become ENFORCED denials +/observe legacy.example.com # discovery allowed, interaction blocked +/guardrail destructive on · accounts 5 · rate 60 +/policy # what is actually enforced +``` + +### Evidence & Validation Engine + +Voting is models checking models, and a confident hallucination passes a vote by +being confident. `crates/harness/src/validation.rs` adds a deterministic layer +that never consults a model: + +``` +HYPOTHESIS → CANDIDATE → [ VALIDATION ENGINE ] → CONFIRMED | NEEDS_REVIEW | REJECTED +``` + +Per-CWE rules, because "is this real?" has a different answer per class: + +| class | what confirms it | +|-------|------------------| +| SQLi (89/943) | baseline vs attack difference **that reproduces ≥2×** | +| XSS (79/80) | a real browser executed a **harness-chosen marker** — reflection alone is not proof | +| IDOR/BOLA (639/862/863) | identity B reads identity A's resource **and the body matches** (a 200 returning a login page is rejected) | +| SSRF (918) | controlled callback, or retrieval of a canary resource | +| LFI (22/23/98) | controlled file marker, or a file signature the baseline lacked | +| RCE (77/78/94) | a unique nonce in command output or a callback — reflected input is rejected | + +Two rules keep it honest: absent evidence is **never** a pass (it becomes +`needs-review`), and a class with no rule is never auto-confirmed. +`NEUROSPLOIT_VALIDATION=advisory|enforcing|off` — advisory (default) rejects +contradictions but won't demote a voted finding merely for missing artifacts; +enforcing makes the verdict the status. + ### Keeping a run going - **Command rectification** — a mistyped command is corrected (`/staus` → `/status`), completed diff --git a/neurosploit-rs/app/src/repl.rs b/neurosploit-rs/app/src/repl.rs index f9272a3..5b6d315 100644 --- a/neurosploit-rs/app/src/repl.rs +++ b/neurosploit-rs/app/src/repl.rs @@ -148,10 +148,11 @@ struct LiveCheckpoint { pub(crate) const ACCEPTED: &[&str] = &[ "/?", "/agents", "/attach", "/auth", "/burp", "/chain", "/changed", "/clear", "/config", "/context", "/continue", "/creds", "/diff", "/exclude", "/exit", "/expand", "/feed", - "/finding", "/findings", "/focus", "/forget", "/full", "/go", "/goal", "/graph", "/help", - "/history", "/idle", "/instructions", "/integration", "/integrations", "/key", "/log", - "/logs", "/mcp", "/memory", "/model", "/models", "/objective", "/objectives", "/offline", - "/onboard", "/only", "/oos", "/outofscope", "/providers", "/proxy", "/q", "/quit", "/recon", + "/finding", "/findings", "/focus", "/forget", "/full", "/go", "/goal", "/graph", "/guardrail", "/guardrails", "/help", + "/history", "/idle", "/inscope", "/instructions", "/integration", "/integrations", "/key", "/log", + "/logs", "/mcp", "/memory", "/model", "/models", "/objective", "/objectives", "/observe", + "/observe-only", "/offline", + "/onboard", "/only", "/oos", "/outofscope", "/policy", "/providers", "/proxy", "/q", "/quit", "/recon", "/repo", "/report", "/results", "/resume", "/retest", "/revalidate", "/run", "/runs", "/scope", "/scope-out", "/show", "/status", "/stop", "/sub", "/subscription", "/target", "/temp-email", "/tempmail", "/theme", "/timeout", "/ua", "/url", "/useragent", "/validate", @@ -164,7 +165,7 @@ const COMMANDS: &[&str] = &[ "/repo", "/auth", "/creds", "/focus", "/objective", "/scope-out", "/attach", "/context", "/mcp", "/offline", "/votes", "/chain", "/recon", "/tempmail", "/timeout", "/proxy", "/burp", "/ua", "/agents", "/only", "/theme", "/clear", "/run", "/stop", "/continue", "/runs", "/results", "/report", "/status", "/logs", "/diff", "/retest", "/validate", "/finding", "/expand", "/integrations", - "/memory", "/forget", "/graph", "/quit", + "/memory", "/forget", "/graph", "/inscope", "/observe", "/guardrail", "/policy", "/quit", ]; /// rustyline helper: Tab-completes `/commands` and `@filesystem-paths`, @@ -280,6 +281,8 @@ struct Session { objective: Option, /// Explicit out-of-scope exclusions the agents must not touch. out_of_scope: Option, + /// Authorization boundary + guardrails, enforced by the harness. + policy: harness::scope::ScopePolicy, attachments: Vec, color: bool, /// Engagement scope from onboarding: web | infra | cloud | ai | skills. @@ -313,6 +316,7 @@ impl Default for Session { instructions: None, objective: None, out_of_scope: None, + policy: Default::default(), attachments: Vec::new(), color: true, scope: "web", @@ -706,7 +710,23 @@ pub async fn repl(base: &Path) -> anyhow::Result<()> { Some(prev) if !prev.trim().is_empty() => format!("{prev}; {arg}"), _ => arg.to_string(), }); - println!(" out-of-scope: {} \x1b[2m(hard constraint — agents skip these)\x1b[0m", s.out_of_scope.clone().unwrap_or_default()); + // Host-shaped entries become ENFORCED exclusions right away, so + // `/policy` shows what will actually be blocked rather than + // deferring the promotion to run time. Prose ("no destructive + // tests") stays prompt guidance — it isn't a pattern. + let mut enforced = 0usize; + for tok in arg.split([',', ';']) { + let t = tok.trim(); + if !t.is_empty() && !t.contains(' ') && (t.contains('.') || t.contains('/')) { + enforced += s.policy.deny(t); + } + } + println!(" out-of-scope: {}", s.out_of_scope.clone().unwrap_or_default()); + if enforced > 0 { + println!(" \x1b[2m{enforced} host rule(s) ENFORCED by the guard — requests there are blocked before they are sent\x1b[0m"); + } else { + println!(" \x1b[2m(guidance for the agents — not a host rule; use /scope-out or /inscope to change the enforced boundary)\x1b[0m"); + } } "/attach" => { let n = attach_path(arg.trim_start_matches('@'), &mut s); if n > 0 { println!(" attached ({} total)", s.attachments.len()); } } "/context" => { @@ -1013,6 +1033,58 @@ pub async fn repl(base: &Path) -> anyhow::Result<()> { } save_session(&s); println!(" session saved → {} · bye.", proj_dir().display()); break; } + "/inscope" | "/policy" => { + if cmd == "/policy" || arg.trim().is_empty() { + let effective = if s.policy.hard.is_empty() { + s.target.as_deref().map(harness::scope::ScopePolicy::for_target) + } else { None }; + let p = effective.as_ref().unwrap_or(&s.policy); + println!(" ┌ scope policy{}", if effective.is_some() { " (derived from /target — nothing added yet)" } else { "" }); + println!(" │ {}", p.summary()); + println!(" └ /inscope · /scope-out · /observe · /guardrail "); + } else { + if s.policy.hard.is_empty() { + // Seed from the target first, or adding one host would + // silently make the target itself out of scope. + if let Some(t) = s.target.clone() { s.policy.allow(&harness::scope::host_of(&t)); } + } + let n = s.policy.allow(arg); + println!(" +{n} in scope · {}", s.policy.summary()); + } + } + "/observe" | "/observe-only" => { + if arg.trim().is_empty() { println!(" usage: /observe — discovery allowed there, interaction blocked"); } + else { let n = s.policy.observe_only(arg); println!(" +{n} observe-only · {}", s.policy.summary()); } + } + "/guardrail" | "/guardrails" => { + let (k, v) = arg.split_once(char::is_whitespace).unwrap_or((arg, "")); + match k.trim().to_lowercase().as_str() { + "" => println!(" guardrails: {} · keys: destructive on|off · accounts · rate ", s.policy.summary()), + "destructive" => { + s.policy.soft.allow_destructive_methods = matches!(v.trim(), "on" | "yes" | "true" | "1"); + println!(" destructive methods: {}", if s.policy.soft.allow_destructive_methods { "ALLOWED" } else { "blocked" }); + } + "accounts" => { + if matches!(v.trim(), "off" | "no" | "0") { + s.policy.soft.allow_account_creation = false; + println!(" account creation: blocked"); + } else { + s.policy.soft.allow_account_creation = true; + let (n, note) = crate::rectify::rectify_count(v, 0, 50, s.policy.soft.max_accounts as usize); + if let Some(note) = note { println!(" \x1b[2m↻ {note}\x1b[0m"); } + s.policy.soft.max_accounts = n as u32; + println!(" account creation: allowed, max {n}"); + } + } + "rate" => { + let (n, note) = crate::rectify::rectify_count(v, 0, 100_000, s.policy.soft.max_requests_per_minute as usize); + if let Some(note) = note { println!(" \x1b[2m↻ {note}\x1b[0m"); } + s.policy.soft.max_requests_per_minute = n as u32; + println!(" rate guard: {} req/min", if n == 0 { "unlimited".into() } else { n.to_string() }); + } + other => println!(" unknown guardrail '{other}' — destructive · accounts · rate"), + } + } "/memory" => memory_cmd(&s, arg), "/forget" => { if arg.trim().is_empty() { @@ -1227,6 +1299,7 @@ async fn run(base: &Path, s: &Session, history: &mut Vec) { }; cfg.objective = s.objective.clone(); cfg.out_of_scope = s.out_of_scope.clone(); + cfg.scope = s.policy.clone(); cfg.auth = s.auth.clone(); cfg.pinned = s.pinned.clone(); // Multiple /auth identities → prepend the access-control (IDOR/BOLA/BFLA) directive. @@ -1302,6 +1375,7 @@ async fn start_background(base: &Path, s: &Session, reader: &mut Reader, else { Some(format!("{}\n\nATTACHED CONTEXT:\n{}", s.instructions.clone().unwrap_or_default(), s.attachments.join("\n\n"))) }; cfg.objective = s.objective.clone(); cfg.out_of_scope = s.out_of_scope.clone(); + cfg.scope = s.policy.clone(); cfg.auth = s.auth.clone(); cfg.pinned = s.pinned.clone(); if matches!(mode_e, crate::Mode::Grey) { cfg.repo = s.repo.clone(); } @@ -1539,6 +1613,16 @@ struct Snapshot { objective: Option, #[serde(default)] out_of_scope: Option, + /// Scope written back as the text the operator typed, so the file stays + /// readable and editable by hand. + #[serde(default)] + scope_in: Vec, + #[serde(default)] + scope_out: Vec, + #[serde(default)] + scope_observe: Vec, + #[serde(default)] + soft: Option, } fn session_path() -> std::path::PathBuf { proj_dir().join("session.json") } fn save_session(s: &Session) { @@ -1548,6 +1632,10 @@ fn save_session(s: &Session) { repo: s.repo.clone(), auth: s.auth.clone(), creds: s.creds.clone(), instructions: s.instructions.clone(), objective: s.objective.clone(), out_of_scope: s.out_of_scope.clone(), + scope_in: s.policy.hard.iter().map(|p| p.as_text()).collect(), + scope_out: s.policy.exclude.iter().map(|p| p.as_text()).collect(), + scope_observe: s.policy.soft.observe_only.iter().map(|p| p.as_text()).collect(), + soft: Some(s.policy.soft.clone()), }; if let Ok(j) = serde_json::to_string_pretty(&snap) { std::fs::write(session_path(), j).ok(); } } @@ -1561,6 +1649,10 @@ fn load_session(s: &mut Session) -> bool { s.target = snap.target; s.repo = snap.repo; s.auth = snap.auth; s.creds = snap.creds; s.instructions = snap.instructions; s.objective = snap.objective; s.out_of_scope = snap.out_of_scope; + if let Some(soft) = snap.soft { s.policy.soft = soft; } + for t in snap.scope_in { s.policy.allow(&t); } + for t in snap.scope_out { s.policy.deny(&t); } + for t in snap.scope_observe { s.policy.observe_only(&t); } true } @@ -1879,7 +1971,11 @@ fn help() { h("/creds ", "creds: jwt/header/cookie/login + ssh/windows + aws/gcp/azure + roles"); h("/focus ", "steer the tests (or just type the instruction)"); h("/objective ", "engagement goal/context — shapes what agents prioritise & count as impact"); - h("/scope-out ", "out-of-scope exclusions — hard constraint, agents skip these (clear to reset)"); + h("/scope-out ", "out-of-scope exclusions — host-shaped entries become ENFORCED denials"); + h("/inscope ","authorize more hosts: host · *.domain · 10.0.0.0/24 · https://host/path"); + h("/observe ", "observe-only: discovery allowed there, interaction blocked"); + h("/guardrail k v", "soft scope: destructive on|off · accounts · rate "); + h("/policy", "show the enforced scope + guardrails"); h("@path @dir @f:1-20", "attach a file/folder/line-range to context (Tab → menu)"); h("/attach ", "attach a file/folder to context"); h("/context", "list current attachments"); diff --git a/neurosploit-rs/crates/harness/src/lib.rs b/neurosploit-rs/crates/harness/src/lib.rs index d73c501..192ead6 100644 --- a/neurosploit-rs/crates/harness/src/lib.rs +++ b/neurosploit-rs/crates/harness/src/lib.rs @@ -22,7 +22,9 @@ pub mod pool; pub mod probe; pub mod report; pub mod rl; +pub mod scope; pub mod types; +pub mod validation; pub use agents::{Agent, Library}; pub use models::{ @@ -34,4 +36,6 @@ pub use pipeline::run; pub use knowledge_graph::{EdgeKind, KnowledgeGraph, NodeKind}; pub use memory::{Memory, Query as MemoryQuery, Tier as MemoryTier}; pub use pool::{ModelPool, Task}; +pub use scope::{Action as ScopeAction, Decision as ScopeDecision, ScopePolicy}; pub use types::{Finding, RunConfig}; +pub use validation::{judge as judge_finding, CweValidator, Evidence, Verdict}; diff --git a/neurosploit-rs/crates/harness/src/pipeline.rs b/neurosploit-rs/crates/harness/src/pipeline.rs index af1ff90..1cc18fe 100644 --- a/neurosploit-rs/crates/harness/src/pipeline.rs +++ b/neurosploit-rs/crates/harness/src/pipeline.rs @@ -41,14 +41,23 @@ fn operator_directives(cfg: &RunConfig) -> String { if let Some(focus) = cfg.instructions.as_deref().filter(|x| !x.trim().is_empty()) { s.push_str(&format!("OPERATOR FOCUS — prioritise this: {focus}\n")); } + // Scope is enforced in code (see `crate::scope`); this block exists so the + // agent does not burn a round trip discovering a boundary the guard would + // have refused anyway. The prose version alone was never a control. + s.push_str(&effective_scope(cfg).prompt_block()); if let Some(oos) = cfg.out_of_scope.as_deref().filter(|x| !x.trim().is_empty()) { s.push_str(&format!( - "OUT OF SCOPE — HARD CONSTRAINT, do NOT test, probe, or interact with any of the following; \ - skip them entirely even if reachable, and never report findings against them: {oos}\n")); + "OUT OF SCOPE — additional operator constraints (techniques, flows, data) beyond the host rules above: {oos}\n")); } if let Some(auth) = cfg.auth.as_deref().filter(|x| !x.trim().is_empty()) { s.push_str(&format!("AUTHENTICATION — test as an authenticated user; send this with each request: {auth}\n")); } + // What the deterministic engine needs in order to confirm a class. Agents + // hold the target *now*; asking for a baseline/attack pair or a marker + // after the run is asking for something that no longer exists. + if crate::validation::Mode::from_env() != crate::validation::Mode::Off { + s.push_str(&crate::validation::evidence_contract()); + } let recalled = memory_directives(cfg); if !recalled.is_empty() { s.push_str(&recalled); @@ -85,6 +94,34 @@ pub(crate) fn run_id(cfg: &RunConfig) -> String { .unwrap_or_default() } +/// The engagement's authorization boundary. +/// +/// Built from the target unless the operator configured one explicitly, with +/// `out_of_scope` entries that name a host or network promoted into real +/// exclusions — until now they were only ever prose in a prompt. +pub fn effective_scope(cfg: &RunConfig) -> crate::scope::ScopePolicy { + let mut p = if cfg.scope.hard.is_empty() { + crate::scope::ScopePolicy::for_target(&cfg.target) + } else { + cfg.scope.clone() + }; + if let Some(repo) = cfg.repo.as_deref().filter(|r| r.starts_with("http")) { + // A grey-box source repo is fetched, not attacked; authorize the fetch. + p.allow(&crate::scope::host_of(repo)); + } + if let Some(oos) = cfg.out_of_scope.as_deref() { + for tok in oos.split([',', ';', '\n']) { + let t = tok.trim(); + // Only host-shaped exclusions become rules; a sentence like "no + // destructive tests" is guidance for the prompt, not a pattern. + if !t.is_empty() && !t.contains(' ') && (t.contains('.') || t.contains('/')) { + p.deny(t); + } + } + } + p +} + /// Prior knowledge about this target, injected into recon/exploit prompts. /// /// An engagement is usually not the first look at a host, but every model call @@ -505,7 +542,7 @@ pub async fn run(cfg: RunConfig, lib: &Library, pool: &ModelPool, tx: Sender = Vec::new(); + let mut kept: Vec = Vec::new(); + for mut f in findings.into_iter() { + match crate::validation::apply_mode(&mut f, vmode) { + Some(crate::validation::Verdict::Confirmed(_)) => { + confirmed += 1; + kept.push(f); + } + Some(crate::validation::Verdict::Rejected(r)) => { + let _ = tx.send(format!("validator rejected '{}': {r}", f.title)).await; + rejected.push(f); + } + _ => kept.push(f), + } + } + findings = kept; + let _ = tx.send(format!( + "validation engine ({:?}): {confirmed} deterministically confirmed, {} rejected, {} kept", + vmode, rejected.len(), findings.len() + )).await; + } + + // Scope audit: anything proven against a host outside the authorization + // boundary is quarantined, not shipped. A finding on an unauthorized asset + // is an incident to disclose, not a deliverable. + let policy = effective_scope(&cfg); + let (in_scope, out_of_scope) = policy.audit_findings(findings); + findings = in_scope; + if !out_of_scope.is_empty() { + let _ = tx.send(format!( + "notify: ⚠ {} finding(s) were proven against hosts OUTSIDE the authorized scope and were withheld: {}", + out_of_scope.len(), + out_of_scope.iter().map(|f| crate::scope::host_of(&f.endpoint)).collect::>().join(", ") + )).await; + if let Some(dir) = cfg.workdir.as_deref() { + let p = Path::new(dir).join("out-of-scope-findings.json"); + let _ = std::fs::write(p, serde_json::to_string_pretty(&out_of_scope).unwrap_or_default()); + } + } + // Durable knowledge. Everything above this point is about *this* run; these // two stores are what makes the next one start from further along. let notes = absorb(&cfg, &recon, &findings); @@ -2157,7 +2239,7 @@ pub async fn run_ai(cfg: RunConfig, lib: &Library, pool: &ModelPool, tx: Sender< // Recon the AI endpoint (probe + model recon). let recon = if cfg.offline { "{}".to_string() } else { - let p = crate::probe::probe(&cfg.target).await; + let p = crate::probe::probe_in_scope(&cfg.target, &effective_scope(&cfg)).await; let _ = tx.send(crate::probe::probe_summary(&p)).await; let facts = crate::probe::probe_json(&p); match pool.complete_routed(Task::Recon, "ai-recon", AI_RECON_SYS, diff --git a/neurosploit-rs/crates/harness/src/probe.rs b/neurosploit-rs/crates/harness/src/probe.rs index f57fa8b..f116b4a 100644 --- a/neurosploit-rs/crates/harness/src/probe.rs +++ b/neurosploit-rs/crates/harness/src/probe.rs @@ -212,6 +212,27 @@ fn parse_forms(body: &str) -> Vec { } /// Run the probe. Never panics; on total failure returns a Probe with a note. +/// Probe a target after checking it against the engagement's boundary. +/// +/// This is the harness's own network chokepoint: it is the one place the +/// harness itself sends requests, so the guard runs here rather than trusting +/// the caller. An out-of-scope target yields an empty probe with a note, not a +/// request. +pub async fn probe_in_scope(target: &str, policy: &crate::scope::ScopePolicy) -> Probe { + let d = policy.check(target, crate::scope::Action::Probe); + if !d.allowed() { + let mut p = Probe::default(); + p.notes.push(format!("scope guard blocked the probe: {}", d.reason())); + return p; + } + if let crate::scope::Decision::Warn(w) = d { + let mut p = probe(target).await; + p.notes.push(format!("scope guard: {w}")); + return p; + } + probe(target).await +} + pub async fn probe(target: &str) -> Probe { let mut p = Probe { url: target.to_string(), ..Default::default() }; let c = client(); diff --git a/neurosploit-rs/crates/harness/src/scope.rs b/neurosploit-rs/crates/harness/src/scope.rs new file mode 100644 index 0000000..4b676fb --- /dev/null +++ b/neurosploit-rs/crates/harness/src/scope.rs @@ -0,0 +1,702 @@ +//! Scope policy guard — hard scope (authorization) and soft scope (guardrails). +//! +//! Before this module, scope existed only as a sentence in the prompt: +//! `out_of_scope` was rendered as "HARD CONSTRAINT — do NOT test…" and nothing +//! checked it. That is a *request*, not a control. An agent that decides a +//! discovered subdomain is interesting, or that follows a redirect off-target, +//! was free to act, and the operator found out by reading the report. In an +//! authorized engagement the boundary is the one thing that must not depend on +//! a model's cooperation. +//! +//! Two layers, because they answer different questions: +//! +//! - **Hard scope** — *are we allowed to touch this at all?* An allowlist of +//! hosts, wildcards, IPv4 CIDRs and URL prefixes, plus exclusions that always +//! win. Anything not matched is [`Decision::Deny`]. It defaults to the +//! engagement's own target, so discovery can never silently widen the +//! engagement: finding a host is not authorization to attack it. +//! - **Soft scope** — *we may touch it, but how?* Guardrails that shape +//! behaviour inside authorized territory: read-only zones, destructive HTTP +//! methods, account creation, request rate, and payload classes that are +//! never acceptable (data destruction, DoS). These produce [`Decision::Warn`] +//! where the action is merely discouraged and [`Decision::Deny`] where it is +//! forbidden. +//! +//! The guard is deterministic and independent of the LLM: [`ScopePolicy::check`] +//! is called at the point of action (probe, validator replay, evidence +//! collection), and [`ScopePolicy::audit_findings`] runs afterwards so anything +//! that reached a host outside the boundary is quarantined instead of shipped +//! in a report. + +use serde::{Deserialize, Serialize}; +use std::collections::VecDeque; +use std::sync::Mutex; + +/// What an interaction intends to do. The same URL can be fine to look at and +/// forbidden to attack, so authorization is per (target, intent), never per +/// target alone. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "kebab-case")] +pub enum Action { + /// Passive: read a page, resolve DNS, look at a response already captured. + Observe, + /// Active but non-mutating: fingerprinting, enumeration, a GET with a probe + /// parameter. + Probe, + /// Sends a payload intended to prove a weakness. + Exploit, + /// Mutates or removes state: DELETE/PUT, account creation, file write. + Destructive, +} + +impl Action { + fn rank(self) -> u8 { + match self { + Action::Observe => 0, + Action::Probe => 1, + Action::Exploit => 2, + Action::Destructive => 3, + } + } +} + +/// The guard's answer. `Warn` still permits the action — it is the honest +/// outcome for "allowed, but the operator should know", and collapsing it into +/// Allow or Deny would either hide the signal or block legitimate testing. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum Decision { + Allow, + Warn(String), + Deny(String), +} + +impl Decision { + pub fn allowed(&self) -> bool { + !matches!(self, Decision::Deny(_)) + } + pub fn reason(&self) -> &str { + match self { + Decision::Allow => "", + Decision::Warn(r) | Decision::Deny(r) => r, + } + } +} + +/// One scope entry. Written by the operator as text and parsed with +/// [`Pattern::parse`], so the config file, the CLI flag and the web form all +/// accept the same spellings. +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "kebab-case", tag = "kind", content = "value")] +pub enum Pattern { + /// Exact host match, case-insensitive (`app.example.com`). + Host(String), + /// Wildcard host (`*.example.com`) — matches sub-domains, and the apex too. + Wildcard(String), + /// IPv4 network (`10.0.0.0/8`). + Cidr { base: u32, bits: u8 }, + /// URL prefix (`https://example.com/api/v2`) — narrower than a whole host. + UrlPrefix(String), +} + +impl Pattern { + pub fn parse(raw: &str) -> Option { + let s = raw.trim().trim_end_matches('.').to_lowercase(); + if s.is_empty() { + return None; + } + if s.contains("://") { + return Some(Pattern::UrlPrefix(s.trim_end_matches('/').to_string())); + } + if let Some((net, bits)) = s.split_once('/') { + if let (Some(base), Ok(bits)) = (ipv4_to_u32(net), bits.parse::()) { + if bits <= 32 { + return Some(Pattern::Cidr { base: base & mask(bits), bits }); + } + } + // A "/" that isn't a CIDR is a path — treat the whole thing as a + // host-relative prefix so `example.com/admin` works as written. + return Some(Pattern::UrlPrefix(format!("https://{s}"))); + } + if let Some(rest) = s.strip_prefix("*.") { + return Some(Pattern::Wildcard(rest.to_string())); + } + Some(Pattern::Host(s)) + } + + pub fn matches(&self, url: &str) -> bool { + let host = host_of(url); + match self { + Pattern::Host(h) => host == *h, + Pattern::Wildcard(root) => host == *root || host.ends_with(&format!(".{root}")), + Pattern::Cidr { base, bits } => ipv4_to_u32(&host).map(|ip| ip & mask(*bits) == *base).unwrap_or(false), + Pattern::UrlPrefix(p) => { + let n = normalize_url(url); + let p = normalize_url(p); + // Prefix on a path boundary: `/api` must not match `/apikeys`. + n == p || n.starts_with(&format!("{p}/")) || n.starts_with(&format!("{p}?")) + } + } + } + + pub fn as_text(&self) -> String { + match self { + Pattern::Host(h) => h.clone(), + Pattern::Wildcard(r) => format!("*.{r}"), + Pattern::Cidr { base, bits } => format!("{}/{}", u32_to_ipv4(*base), bits), + Pattern::UrlPrefix(p) => p.clone(), + } + } +} + +fn mask(bits: u8) -> u32 { + if bits == 0 { + 0 + } else { + u32::MAX << (32 - bits.min(32)) + } +} + +fn ipv4_to_u32(s: &str) -> Option { + let parts: Vec<&str> = s.split('.').collect(); + if parts.len() != 4 { + return None; + } + let mut out: u32 = 0; + for p in parts { + let n: u32 = p.parse().ok()?; + if n > 255 { + return None; + } + out = (out << 8) | n; + } + Some(out) +} + +fn u32_to_ipv4(v: u32) -> String { + format!("{}.{}.{}.{}", v >> 24, (v >> 16) & 255, (v >> 8) & 255, v & 255) +} + +/// Host of a URL or bare authority, lowercased, without port or userinfo. +pub fn host_of(url: &str) -> String { + let s = url.trim().to_lowercase(); + let s = s.split_once("://").map(|(_, r)| r).unwrap_or(&s); + let s = s.split(['/', '?', '#']).next().unwrap_or(s); + let s = s.rsplit_once('@').map(|(_, h)| h).unwrap_or(s); + // IPv6 literals keep their brackets; for anything else a colon is a port. + if s.starts_with('[') { + return s.split(']').next().unwrap_or(s).trim_start_matches('[').to_string(); + } + s.split(':').next().unwrap_or(s).trim_start_matches("www.").to_string() +} + +fn normalize_url(url: &str) -> String { + let s = url.trim().to_lowercase(); + let s = s.split_once("://").map(|(_, r)| r).unwrap_or(&s); + s.trim_end_matches('/').trim_start_matches("www.").to_string() +} + +/// Guardrails that apply *inside* authorized scope. +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct SoftScope { + /// Hosts/prefixes that may be looked at but never attacked. + #[serde(default)] + pub observe_only: Vec, + /// Allow DELETE/PUT/PATCH and other state-mutating verbs. + #[serde(default)] + pub allow_destructive_methods: bool, + /// Allow the agent to register test accounts. + #[serde(default = "yes")] + pub allow_account_creation: bool, + /// Cap on accounts created during the engagement (0 = unlimited). + #[serde(default = "default_max_accounts")] + pub max_accounts: u32, + /// Requests per minute across the engagement (0 = unlimited). + #[serde(default = "default_rate")] + pub max_requests_per_minute: u32, + /// Payload substrings that are never acceptable, whatever the finding. + /// Defaults cover data destruction and resource exhaustion — the two + /// classes that damage a production target rather than demonstrate a bug. + #[serde(default = "default_forbidden_payloads")] + pub forbidden_payloads: Vec, + /// Free-text notes from the operator, passed to prompts as context. Not + /// enforceable — kept separate from the rules precisely so nobody mistakes + /// prose for a control. + #[serde(default)] + pub notes: Vec, +} + +fn yes() -> bool { + true +} +fn default_max_accounts() -> u32 { + 3 +} +fn default_rate() -> u32 { + 240 +} +fn default_forbidden_payloads() -> Vec { + [ + "drop table", "drop database", "truncate table", "delete from users", + "rm -rf /", "mkfs", "shutdown -h", "format c:", ":(){:|:&};:", + "while(true)", "sleep(100)", "benchmark(10000000", + ] + .iter() + .map(|s| s.to_string()) + .collect() +} + +impl Default for SoftScope { + fn default() -> Self { + SoftScope { + observe_only: Vec::new(), + allow_destructive_methods: false, + allow_account_creation: true, + max_accounts: default_max_accounts(), + max_requests_per_minute: default_rate(), + forbidden_payloads: default_forbidden_payloads(), + notes: Vec::new(), + } + } +} + +/// The engagement's authorization boundary plus its guardrails. +#[derive(Debug, Default, Serialize, Deserialize)] +pub struct ScopePolicy { + /// Allowlist. Empty means "nothing is authorized" — see [`ScopePolicy::for_target`]. + #[serde(default)] + pub hard: Vec, + /// Exclusions. Always beat the allowlist. + #[serde(default)] + pub exclude: Vec, + #[serde(default)] + pub soft: SoftScope, + /// Rolling request timestamps for the rate guard. + #[serde(skip)] + ledger: Mutex>, + #[serde(skip)] + accounts: std::sync::atomic::AtomicU32, +} + +impl Clone for ScopePolicy { + fn clone(&self) -> Self { + ScopePolicy { + hard: self.hard.clone(), + exclude: self.exclude.clone(), + soft: self.soft.clone(), + ledger: Mutex::new(VecDeque::new()), + accounts: std::sync::atomic::AtomicU32::new(self.accounts.load(std::sync::atomic::Ordering::Relaxed)), + } + } +} + +impl ScopePolicy { + /// The default policy for an engagement: exactly the target, nothing else. + /// + /// Starting from the target rather than from "everything" is the whole + /// point. Recon finds subdomains, third-party CDNs, SSO providers and + /// internal hosts referenced in JavaScript; none of that is authorized, and + /// an agent that treats discovery as permission is how an engagement ends + /// up touching someone else's asset. + pub fn for_target(target: &str) -> ScopePolicy { + let mut p = ScopePolicy::default(); + if let Some(pat) = Pattern::parse(&host_of(target)) { + p.hard.push(pat); + } + p + } + + /// Add allowlist entries from operator text (comma/space/newline separated). + pub fn allow(&mut self, raw: &str) -> usize { + let before = self.hard.len(); + for tok in split_list(raw) { + if let Some(p) = Pattern::parse(&tok) { + if !self.hard.contains(&p) { + self.hard.push(p); + } + } + } + self.hard.len() - before + } + + /// Add exclusions from operator text. Exclusions beat the allowlist, so an + /// operator can authorize `*.example.com` and still carve out `payments.`. + pub fn deny(&mut self, raw: &str) -> usize { + let before = self.exclude.len(); + for tok in split_list(raw) { + if let Some(p) = Pattern::parse(&tok) { + if !self.exclude.contains(&p) { + self.exclude.push(p); + } + } + } + self.exclude.len() - before + } + + /// Mark hosts/prefixes as look-but-don't-touch. + pub fn observe_only(&mut self, raw: &str) -> usize { + let before = self.soft.observe_only.len(); + for tok in split_list(raw) { + if let Some(p) = Pattern::parse(&tok) { + if !self.soft.observe_only.contains(&p) { + self.soft.observe_only.push(p); + } + } + } + self.soft.observe_only.len() - before + } + + pub fn in_hard_scope(&self, url: &str) -> bool { + if self.exclude.iter().any(|p| p.matches(url)) { + return false; + } + self.hard.iter().any(|p| p.matches(url)) + } + + /// The authorization decision for one interaction. + pub fn check(&self, url: &str, action: Action) -> Decision { + let host = host_of(url); + if host.is_empty() { + return Decision::Deny("no host in target".into()); + } + if let Some(p) = self.exclude.iter().find(|p| p.matches(url)) { + return Decision::Deny(format!("{host} is excluded by scope rule '{}'", p.as_text())); + } + if self.hard.is_empty() { + return Decision::Deny("no hard scope configured — nothing is authorized".into()); + } + if !self.hard.iter().any(|p| p.matches(url)) { + return Decision::Deny(format!( + "{host} is outside the authorized scope ({})", + self.hard.iter().map(|p| p.as_text()).collect::>().join(", ") + )); + } + if action.rank() >= Action::Exploit.rank() { + if let Some(p) = self.soft.observe_only.iter().find(|p| p.matches(url)) { + return Decision::Deny(format!("{} is observe-only — discovery allowed, interaction is not", p.as_text())); + } + } + if action == Action::Destructive && !self.soft.allow_destructive_methods { + return Decision::Deny("destructive actions are disabled for this engagement".into()); + } + if let Some(w) = self.rate_check() { + return w; + } + Decision::Allow + } + + /// Check an outbound HTTP request: the URL, the verb and the payload. + pub fn check_request(&self, url: &str, method: &str, body: &str) -> Decision { + let m = method.trim().to_uppercase(); + let action = match m.as_str() { + "GET" | "HEAD" | "OPTIONS" => Action::Probe, + "DELETE" | "PUT" | "PATCH" => Action::Destructive, + _ => Action::Exploit, + }; + if let Some(bad) = self.forbidden_in(body).or_else(|| self.forbidden_in(url)) { + return Decision::Deny(format!("payload contains a forbidden pattern ('{bad}') — this damages the target instead of proving a bug")); + } + self.check(url, action) + } + + fn forbidden_in(&self, s: &str) -> Option { + if s.is_empty() { + return None; + } + let hay = s.to_lowercase(); + self.soft.forbidden_payloads.iter().find(|p| hay.contains(&p.to_lowercase())).cloned() + } + + /// Account creation is capped rather than forbidden: a test account is + /// often the only way to prove an access-control bug, but an agent looping + /// on a registration form is abuse. + pub fn check_account_creation(&self) -> Decision { + if !self.soft.allow_account_creation { + return Decision::Deny("account creation is disabled for this engagement".into()); + } + let n = self.accounts.load(std::sync::atomic::Ordering::Relaxed); + if self.soft.max_accounts > 0 && n >= self.soft.max_accounts { + return Decision::Deny(format!("account cap reached ({} of {})", n, self.soft.max_accounts)); + } + Decision::Allow + } + + pub fn note_account_created(&self) { + self.accounts.fetch_add(1, std::sync::atomic::Ordering::Relaxed); + } + + /// Record a request and report whether the rate guard is tripped. + fn rate_check(&self) -> Option { + if self.soft.max_requests_per_minute == 0 { + return None; + } + let now = std::time::Instant::now(); + let mut led = self.ledger.lock().ok()?; + while led.front().map(|t| now.duration_since(*t).as_secs() >= 60).unwrap_or(false) { + led.pop_front(); + } + led.push_back(now); + if led.len() as u32 > self.soft.max_requests_per_minute { + // A warning, not a denial: the engagement should slow down, and + // silently dropping a request would make the agent misread the + // target as unreachable. + return Some(Decision::Warn(format!( + "request rate above {} per minute — throttle", + self.soft.max_requests_per_minute + ))); + } + None + } + + /// Quarantine findings proven against something outside the boundary. + /// + /// Returns `(kept, quarantined)`. A finding on an unauthorized host is not + /// a finding to report — it is an incident to disclose to the operator, and + /// shipping it in the deliverable would launder the mistake. + pub fn audit_findings(&self, findings: Vec) -> (Vec, Vec) { + let mut kept = Vec::new(); + let mut out = Vec::new(); + for f in findings { + // A finding with no endpoint (many SAST results) has no host to + // check; source review is bounded by the repo, not by the network. + if f.endpoint.trim().is_empty() + || looks_like_source_ref(&f.endpoint) + || host_of(&f.endpoint).is_empty() + || self.in_hard_scope(&f.endpoint) + { + kept.push(f); + } else { + out.push(f); + } + } + (kept, out) + } + + /// The scope block rendered for prompts. The rules are enforced in code; + /// this exists so the agent does not waste a round trip discovering a + /// boundary the guard would have refused anyway. + pub fn prompt_block(&self) -> String { + if self.hard.is_empty() { + return String::new(); + } + let mut s = String::from("AUTHORIZED SCOPE — enforced by the harness, not advisory. Requests outside it are blocked before they are sent:\n"); + s.push_str(&format!(" in scope: {}\n", self.hard.iter().map(|p| p.as_text()).collect::>().join(", "))); + if !self.exclude.is_empty() { + s.push_str(&format!(" excluded: {}\n", self.exclude.iter().map(|p| p.as_text()).collect::>().join(", "))); + } + if !self.soft.observe_only.is_empty() { + s.push_str(&format!(" observe-only (look, never interact): {}\n", self.soft.observe_only.iter().map(|p| p.as_text()).collect::>().join(", "))); + } + s.push_str(&format!( + " guardrails: destructive methods {}, account creation {}{}, max {} req/min\n", + if self.soft.allow_destructive_methods { "ALLOWED" } else { "BLOCKED" }, + if self.soft.allow_account_creation { "allowed" } else { "BLOCKED" }, + if self.soft.allow_account_creation && self.soft.max_accounts > 0 { format!(" (max {})", self.soft.max_accounts) } else { String::new() }, + self.soft.max_requests_per_minute + )); + s.push_str(" Discovering a host, link, subdomain or API does NOT authorize testing it. Report it as an observation instead.\n"); + for n in &self.soft.notes { + s.push_str(&format!(" note: {n}\n")); + } + s + } + + /// One-line summary for `/scope` and the web console. + pub fn summary(&self) -> String { + format!( + "hard: {} · excluded: {} · observe-only: {} · destructive: {} · accounts: {} · {} req/min", + if self.hard.is_empty() { "(none — nothing authorized)".into() } else { self.hard.iter().map(|p| p.as_text()).collect::>().join(",") }, + if self.exclude.is_empty() { "-".into() } else { self.exclude.iter().map(|p| p.as_text()).collect::>().join(",") }, + if self.soft.observe_only.is_empty() { "-".into() } else { self.soft.observe_only.iter().map(|p| p.as_text()).collect::>().join(",") }, + if self.soft.allow_destructive_methods { "allowed" } else { "blocked" }, + if self.soft.allow_account_creation { format!("max {}", self.soft.max_accounts) } else { "blocked".into() }, + self.soft.max_requests_per_minute + ) + } +} + +/// Does this endpoint name a place in source rather than a place on the +/// network? SAST findings carry `file.ext:line`, which has no host to +/// authorize — source review is bounded by the repository, not by scope. The +/// distinction matters because `example.com:8080` also ends in `:digits`, so +/// the discriminator is a path separator or a known code extension, not the +/// colon. +pub fn looks_like_source_ref(s: &str) -> bool { + let s = s.trim(); + if s.is_empty() || s.contains("://") { + return false; + } + if s.starts_with('/') || s.starts_with("./") || s.starts_with("../") { + return true; + } + const CODE_EXT: &[&str] = &[ + "rs", "js", "mjs", "cjs", "ts", "tsx", "jsx", "vue", "py", "java", "kt", "go", "rb", "php", + "c", "h", "cc", "cpp", "hpp", "cs", "swift", "scala", "sql", "sh", "bash", "ps1", "tf", + "yaml", "yml", "json", "toml", "xml", "md", "erb", "ejs", "twig", "jsp", "aspx", "cshtml", + ]; + let Some((path, tail)) = s.rsplit_once(':') else { return false }; + if tail.is_empty() || !tail.chars().all(|c| c.is_ascii_digit()) { + return false; + } + let base = path.rsplit(['/', '\\']).next().unwrap_or(path); + let ext = base.rsplit_once('.').map(|(_, e)| e.to_lowercase()).unwrap_or_default(); + path.contains('/') || path.contains('\\') || CODE_EXT.contains(&ext.as_str()) +} + +fn split_list(raw: &str) -> Vec { + raw.split([',', ';', ' ', '\n', '\t']) + .map(|s| s.trim().to_string()) + .filter(|s| !s.is_empty()) + .collect() +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::types::Finding; + + fn policy() -> ScopePolicy { + let mut p = ScopePolicy::for_target("https://app.example.com/login"); + p.soft.max_requests_per_minute = 0; // rate guard tested separately + p + } + + #[test] + fn the_default_scope_is_the_target_and_nothing_else() { + let p = policy(); + assert_eq!(p.check("https://app.example.com/admin", Action::Exploit), Decision::Allow); + // Discovery is not authorization: a subdomain found in recon stays out. + match p.check("https://internal.example.com/", Action::Probe) { + Decision::Deny(r) => assert!(r.contains("outside the authorized scope"), "{r}"), + d => panic!("a discovered subdomain must not be authorized: {d:?}"), + } + assert!(!p.check("https://cdn.thirdparty.net/app.js", Action::Observe).allowed()); + } + + #[test] + fn a_wildcard_covers_subdomains_and_the_apex() { + let mut p = policy(); + p.allow("*.example.com"); + assert!(p.check("https://api.example.com/v1", Action::Exploit).allowed()); + assert!(p.check("https://example.com/", Action::Exploit).allowed()); + assert!(!p.check("https://example.com.evil.net/", Action::Probe).allowed(), "suffix confusion must not pass"); + } + + #[test] + fn an_exclusion_beats_the_allowlist() { + let mut p = policy(); + p.allow("*.example.com"); + p.deny("payments.example.com"); + match p.check("https://payments.example.com/checkout", Action::Probe) { + Decision::Deny(r) => assert!(r.contains("excluded"), "{r}"), + d => panic!("exclusion must win over the wildcard: {d:?}"), + } + } + + #[test] + fn cidr_scope_matches_addresses_in_the_network_only() { + let mut p = ScopePolicy::default(); + p.soft.max_requests_per_minute = 0; + p.allow("10.0.0.0/24"); + assert!(p.check("http://10.0.0.7:8080/", Action::Exploit).allowed()); + assert!(!p.check("http://10.0.1.7/", Action::Probe).allowed()); + } + + #[test] + fn a_url_prefix_scopes_one_path_not_its_neighbours() { + let mut p = ScopePolicy::default(); + p.soft.max_requests_per_minute = 0; + p.allow("https://example.com/api"); + assert!(p.check("https://example.com/api/users", Action::Exploit).allowed()); + assert!(!p.check("https://example.com/apikeys", Action::Probe).allowed(), "/api must not match /apikeys"); + assert!(!p.check("https://example.com/admin", Action::Probe).allowed()); + } + + #[test] + fn observe_only_permits_looking_and_refuses_touching() { + let mut p = policy(); + p.allow("*.example.com"); + p.observe_only("legacy.example.com"); + assert!(p.check("https://legacy.example.com/", Action::Observe).allowed()); + assert!(p.check("https://legacy.example.com/", Action::Probe).allowed()); + assert!(!p.check("https://legacy.example.com/", Action::Exploit).allowed()); + } + + #[test] + fn destructive_verbs_are_off_until_the_operator_turns_them_on() { + let mut p = policy(); + assert!(!p.check_request("https://app.example.com/orders/1", "DELETE", "").allowed()); + p.soft.allow_destructive_methods = true; + assert!(p.check_request("https://app.example.com/orders/1", "DELETE", "").allowed()); + } + + #[test] + fn payloads_that_destroy_data_are_refused_even_in_scope() { + let p = policy(); + match p.check_request("https://app.example.com/search", "POST", "q=1'; DROP TABLE users--") { + Decision::Deny(r) => assert!(r.contains("forbidden pattern"), "{r}"), + d => panic!("a destructive payload must be refused: {d:?}"), + } + // The benign equivalent of the same test still goes through. + assert!(p.check_request("https://app.example.com/search", "POST", "q=1' OR '1'='1").allowed()); + } + + #[test] + fn account_creation_is_capped_not_unlimited() { + let p = policy(); + for _ in 0..p.soft.max_accounts { + assert!(p.check_account_creation().allowed()); + p.note_account_created(); + } + match p.check_account_creation() { + Decision::Deny(r) => assert!(r.contains("cap reached"), "{r}"), + d => panic!("the cap must hold: {d:?}"), + } + } + + #[test] + fn the_rate_guard_warns_without_blocking() { + let mut p = policy(); + p.soft.max_requests_per_minute = 2; + assert_eq!(p.check("https://app.example.com/a", Action::Probe), Decision::Allow); + assert_eq!(p.check("https://app.example.com/b", Action::Probe), Decision::Allow); + match p.check("https://app.example.com/c", Action::Probe) { + // Still allowed — dropping it would read as "target unreachable". + Decision::Warn(r) => assert!(r.contains("request rate")), + d => panic!("expected a throttle warning, got {d:?}"), + } + } + + #[test] + fn findings_proven_outside_the_boundary_are_quarantined() { + let p = policy(); + let inside = Finding { endpoint: "https://app.example.com/login".into(), title: "in".into(), ..Default::default() }; + let outside = Finding { endpoint: "https://other.test/x".into(), title: "out".into(), ..Default::default() }; + let sast = Finding { endpoint: "src/auth.rs:42".into(), title: "code".into(), ..Default::default() }; + let (kept, quarantined) = p.audit_findings(vec![inside, outside, sast]); + assert_eq!(kept.iter().map(|f| f.title.clone()).collect::>(), vec!["in", "code"]); + assert_eq!(quarantined.len(), 1); + assert_eq!(quarantined[0].title, "out"); + } + + #[test] + fn a_source_reference_is_not_a_host() { + assert!(looks_like_source_ref("src/auth.rs:42")); + assert!(looks_like_source_ref("app/Main.java:100")); + assert!(looks_like_source_ref("auth.rs:7"), "a bare file with a code extension still counts"); + assert!(looks_like_source_ref("/etc/passwd")); + // The shape that made this necessary: a host with a port ends in + // :digits too, and must stay a network endpoint. + assert!(!looks_like_source_ref("example.com:8080")); + assert!(!looks_like_source_ref("https://example.com/a.rs:42")); + assert!(!looks_like_source_ref("10.0.0.1:22")); + } + + #[test] + fn an_empty_policy_authorizes_nothing() { + let p = ScopePolicy::default(); + match p.check("https://anything.test/", Action::Observe) { + Decision::Deny(r) => assert!(r.contains("nothing is authorized"), "{r}"), + d => panic!("an unconfigured policy must be closed, not open: {d:?}"), + } + } +} diff --git a/neurosploit-rs/crates/harness/src/types.rs b/neurosploit-rs/crates/harness/src/types.rs index a704ed7..eaf57f5 100644 --- a/neurosploit-rs/crates/harness/src/types.rs +++ b/neurosploit-rs/crates/harness/src/types.rs @@ -77,6 +77,12 @@ pub struct Finding { /// report can embed each image next to its vulnerability. #[serde(default)] pub screenshots: Vec, + /// Structured artifacts for the deterministic validation engine: the + /// baseline/attack pair, repeats, markers, identity pair. Agents that + /// follow the evidence contract emit this alongside the finding, and + /// `crate::validation` judges the class from it without consulting a model. + #[serde(default)] + pub evidence_data: Option, } impl Default for Finding { @@ -108,6 +114,7 @@ impl Default for Finding { review_status: String::new(), review_reason: String::new(), screenshots: Vec::new(), + evidence_data: None, } } } @@ -197,6 +204,11 @@ pub struct RunConfig { /// the app to `/.neurosploit/vault`; falls back to the run workdir. #[serde(default)] pub vault_dir: Option, + /// Authorization boundary and guardrails. Empty `hard` means "derive from + /// the target" (see `pipeline::effective_scope`) — an engagement is never + /// implicitly authorized against anything but what it was pointed at. + #[serde(default)] + pub scope: crate::scope::ScopePolicy, } fn default_vote() -> usize { @@ -239,6 +251,7 @@ impl RunConfig { recon_intensity: 3, temp_email: false, vault_dir: None, + scope: Default::default(), } } } diff --git a/neurosploit-rs/crates/harness/src/validation.rs b/neurosploit-rs/crates/harness/src/validation.rs new file mode 100644 index 0000000..594f482 --- /dev/null +++ b/neurosploit-rs/crates/harness/src/validation.rs @@ -0,0 +1,867 @@ +//! Evidence & Validation Engine — deterministic, per-CWE, outside the model. +//! +//! Today a finding becomes "validated" by asking more language models: N-model +//! voting, then an adversarial refute pass. That catches sloppy reasoning, but +//! it shares the failure mode of the thing it checks — models agreeing with +//! each other is not evidence, and a confident hallucination survives a vote by +//! being confident. [`crate::grounding`] adds a receipt requirement, but it +//! matches keywords ("http/", "status", "alert(") and cannot tell a real +//! response apart from a plausible transcript of one. +//! +//! This engine asks a different question: **does the recorded evidence actually +//! demonstrate this specific weakness?** The rule is per-CWE because the answer +//! is: SQL injection is proven by a reproducible, deterministic difference +//! between a baseline and an attack response; XSS is proven by a browser +//! executing a marker the harness chose; IDOR is proven by identity B reading +//! identity A's resource *and* the response carrying A's data. None of those +//! reduce to "the evidence looks technical". +//! +//! ```text +//! HYPOTHESIS an agent noticed something +//! │ +//! CANDIDATE a reproducible interaction was built +//! │ +//! ┌────┴──────────────────────┐ +//! │ VALIDATION ENGINE │ deterministic, per-CWE, no LLM +//! └────┬──────────────────────┘ +//! PASS │ UNCERTAIN │ FAIL +//! ▼ ▼ ▼ +//! CONFIRMED NEEDS_REVIEW REJECTED +//! ``` +//! +//! Two rules keep the engine honest: +//! +//! 1. **Absent evidence is never a pass.** A class with no validator, or a +//! finding whose evidence was never captured, lands in `NeedsReview` — the +//! engine says "I could not prove this", never "this is fine". +//! 2. **It can refuse, and it can confirm, but it cannot invent.** A verdict is +//! a function of recorded artifacts. Nothing here consults a model. + +use crate::types::Finding; +use serde::{Deserialize, Serialize}; + +/// One recorded HTTP interaction. Deliberately small: what a validator needs is +/// what distinguishes two responses, not a full transcript. +#[derive(Debug, Clone, Default, Serialize, Deserialize)] +pub struct Exchange { + pub method: String, + pub url: String, + pub status: u16, + pub body: String, + pub content_type: String, + pub elapsed_ms: u64, + /// Identity this exchange was performed as ("", "userA", "admin", …). + #[serde(default)] + pub identity: String, +} + +impl Exchange { + pub fn len(&self) -> usize { + self.body.len() + } + pub fn is_empty(&self) -> bool { + self.body.is_empty() + } +} + +/// Everything the engine may reason about for one candidate finding. +#[derive(Debug, Clone, Default, Serialize, Deserialize)] +pub struct Evidence { + /// The same request without the payload — what "normal" looks like. + pub baseline: Option, + /// The request carrying the payload. + pub attack: Option, + /// Independent repeats of the attack, for reproducibility. + #[serde(default)] + pub repeats: Vec, + /// A token the harness generated, so observing it cannot be a coincidence. + #[serde(default)] + pub marker: String, + /// The marker was observed where it proves the class (rendered DOM, file + /// read-back, command output, callback). + #[serde(default)] + pub marker_observed: bool, + /// A real browser executed the payload (XSS), not a string match in HTML. + #[serde(default)] + pub browser_executed: bool, + /// An out-of-band callback carrying the marker was received (SSRF, blind RCE). + #[serde(default)] + pub callback_received: bool, + /// Access-control pairs: the resource as its owner, and the same resource + /// requested by a different identity. + pub identity_a: Option, + pub identity_b: Option, + #[serde(default)] + pub notes: Vec, +} + +/// What the engine concluded. +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "kebab-case", tag = "verdict", content = "reason")] +pub enum Verdict { + /// The evidence demonstrates this class. Report it. + Confirmed(String), + /// Could not be proven either way — a human decides. This is the default + /// for anything the engine does not have a rule for. + NeedsReview(String), + /// The evidence contradicts the claim. Drop it to informational. + Rejected(String), +} + +impl Verdict { + pub fn status(&self) -> &'static str { + match self { + Verdict::Confirmed(_) => "confirmed", + Verdict::NeedsReview(_) => "needs-review", + Verdict::Rejected(_) => "rejected", + } + } + pub fn reason(&self) -> &str { + match self { + Verdict::Confirmed(r) | Verdict::NeedsReview(r) | Verdict::Rejected(r) => r, + } + } +} + +/// Measured difference between a baseline and an attack response. +#[derive(Debug, Clone, Default, PartialEq)] +pub struct Diff { + pub status_changed: bool, + pub baseline_status: u16, + pub attack_status: u16, + /// Relative length change, 0.0..1.0+. + pub len_ratio: f64, + pub len_delta: i64, + /// Milliseconds slower (negative = faster). + pub timing_delta_ms: i64, + /// A database/interpreter error surfaced only under attack. + pub error_signature: Option, +} + +/// Error strings that indicate the payload reached an interpreter. Matched +/// against the attack response only when the baseline did NOT contain them — +/// an app that always prints SQL errors proves nothing about this payload. +const DB_ERRORS: &[&str] = &[ + "sql syntax", "mysql_fetch", "mysqli", "ora-01756", "ora-00933", "psql:", "pg_query", + "sqlite3::", "sqlstate", "unclosed quotation mark", "quoted string not properly terminated", + "odbc microsoft access", "microsoft ole db", "incorrect syntax near", "invalid sql statement", + "postgresql query failed", "supplied argument is not a valid mysql", +]; + +pub fn diff(baseline: &Exchange, attack: &Exchange) -> Diff { + let b_len = baseline.len() as i64; + let a_len = attack.len() as i64; + let len_ratio = if b_len == 0 { if a_len == 0 { 1.0 } else { f64::INFINITY } } else { a_len as f64 / b_len as f64 }; + let blow = baseline.body.to_lowercase(); + let alow = attack.body.to_lowercase(); + let error_signature = DB_ERRORS + .iter() + .find(|e| alow.contains(**e) && !blow.contains(**e)) + .map(|e| (*e).to_string()); + Diff { + status_changed: baseline.status != attack.status, + baseline_status: baseline.status, + attack_status: attack.status, + len_ratio, + len_delta: a_len - b_len, + timing_delta_ms: attack.elapsed_ms as i64 - baseline.elapsed_ms as i64, + error_signature, + } +} + +impl Diff { + /// Is this difference big enough to mean something? Small jitter in a + /// dynamic page (timestamps, CSRF tokens, ads) is normal, so the threshold + /// sits above it deliberately. + pub fn is_significant(&self) -> bool { + self.error_signature.is_some() + || self.status_changed + || self.len_ratio.is_infinite() + || (self.len_ratio - 1.0).abs() >= 0.10 + || self.timing_delta_ms >= 4000 + } + + pub fn describe(&self) -> String { + if let Some(e) = &self.error_signature { + return format!("interpreter error '{e}' appeared only under the payload"); + } + if self.status_changed { + return format!("status {} → {}", self.baseline_status, self.attack_status); + } + if self.timing_delta_ms >= 4000 { + return format!("response {}ms slower under the payload", self.timing_delta_ms); + } + format!("body length {:+} bytes ({:.0}% of baseline)", self.len_delta, self.len_ratio * 100.0) + } +} + +/// A canary the harness chose. Observing it in the right place cannot be a +/// coincidence, which is the difference between evidence and a string that +/// looked suspicious. +pub fn canary(prefix: &str) -> String { + // The clock alone is not enough: two canaries minted inside the same tick + // came out identical, and a marker that repeats proves nothing — it could + // have come from the previous test. A process-wide counter makes + // uniqueness independent of clock resolution, and the pid keeps two + // concurrent runs from colliding. + static SEQ: std::sync::atomic::AtomicU64 = std::sync::atomic::AtomicU64::new(0); + let seq = SEQ.fetch_add(1, std::sync::atomic::Ordering::Relaxed); + let n = std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .map(|d| d.as_nanos() as u64) + .unwrap_or(0); + // Not cryptographic — it only has to be unguessable enough that the target + // could not have produced it on its own. + let mut h: u64 = 0xcbf2_9ce4_8422_2325 ^ n; + h ^= seq.wrapping_mul(0x9e37_79b9_7f4a_7c15); + h ^= (std::process::id() as u64).wrapping_mul(0xbf58_476d_1ce4_e5b9); + h = h.wrapping_mul(0x1000_0000_01b3); + h ^= h >> 29; + h = h.wrapping_mul(0xff51_afd7_ed55_8ccd); + h ^= h >> 32; + format!("{prefix}{:012x}", h & 0xffff_ffff_ffff) +} + +/// Did every repeat reproduce the same significant difference? One occurrence +/// of a length change on a dynamic page is noise; the same change three times +/// is behaviour. +pub fn reproducible(baseline: &Exchange, repeats: &[Exchange], min: usize) -> (bool, usize) { + let hits = repeats.iter().filter(|r| diff(baseline, r).is_significant()).count(); + (hits >= min, hits) +} + +/// A per-CWE rule. Each answers one question about recorded artifacts. +pub trait CweValidator: Send + Sync { + fn name(&self) -> &'static str; + /// CWE ids (bare numbers) this rule owns. + fn cwes(&self) -> &'static [&'static str]; + /// What the class needs before it can be confirmed — shown to the operator + /// and to the agent, so "what would prove this" is never a guess. + fn evidence_required(&self) -> &'static [&'static str]; + fn validate(&self, f: &Finding, ev: &Evidence) -> Verdict; +} + +fn cwe_num(cwe: &str) -> String { + cwe.chars().filter(|c| c.is_ascii_digit()).collect() +} + +pub struct SqliValidator; +impl CweValidator for SqliValidator { + fn name(&self) -> &'static str { + "sqli" + } + fn cwes(&self) -> &'static [&'static str] { + &["89", "943", "564"] + } + fn evidence_required(&self) -> &'static [&'static str] { + &["baseline_request", "attack_request", "deterministic_behavior_difference", "reproducibility >= 2"] + } + fn validate(&self, _f: &Finding, ev: &Evidence) -> Verdict { + let (Some(b), Some(a)) = (&ev.baseline, &ev.attack) else { + return Verdict::NeedsReview("no baseline/attack pair was captured — injection cannot be judged from a single response".into()); + }; + let d = diff(b, a); + if !d.is_significant() { + return Verdict::Rejected(format!("payload changed nothing measurable ({})", d.describe())); + } + // Reproducibility is the whole point: a one-off difference on a dynamic + // page is the most common false positive in this class. + let (ok, hits) = reproducible(b, &ev.repeats, 2); + if !ok { + return Verdict::NeedsReview(format!( + "difference observed ({}) but reproduced only {hits}/{} times — not deterministic", + d.describe(), + ev.repeats.len() + )); + } + Verdict::Confirmed(format!("{} — reproduced {hits}/{}", d.describe(), ev.repeats.len())) + } +} + +pub struct XssValidator; +impl CweValidator for XssValidator { + fn name(&self) -> &'static str { + "xss" + } + fn cwes(&self) -> &'static [&'static str] { + &["79", "80", "83", "87"] + } + fn evidence_required(&self) -> &'static [&'static str] { + &["browser_execution", "controlled_marker", "DOM/runtime confirmation"] + } + fn validate(&self, _f: &Finding, ev: &Evidence) -> Verdict { + if ev.marker.is_empty() { + return Verdict::NeedsReview("no controlled marker — a payload echoed in HTML is reflection, not proof of execution".into()); + } + if !ev.browser_executed { + let reflected = ev + .attack + .as_ref() + .map(|a| a.body.contains(&ev.marker)) + .unwrap_or(false); + return if reflected { + Verdict::NeedsReview("marker is reflected but no browser executed it — could be encoded, CSP-blocked, or in a non-executing context".into()) + } else { + Verdict::Rejected("marker never reached the response".into()) + }; + } + if !ev.marker_observed { + return Verdict::NeedsReview("browser ran but the marker was not observed at runtime".into()); + } + Verdict::Confirmed(format!("browser executed the payload and reported marker {}", ev.marker)) + } +} + +pub struct IdorValidator; +impl CweValidator for IdorValidator { + fn name(&self) -> &'static str { + "idor" + } + fn cwes(&self) -> &'static [&'static str] { + &["639", "862", "863", "284", "285", "566", "425"] + } + fn evidence_required(&self) -> &'static [&'static str] { + &["identity_A_resource", "identity_B_request", "successful unauthorized access", "response_semantics_match"] + } + fn validate(&self, _f: &Finding, ev: &Evidence) -> Verdict { + let (Some(a), Some(b)) = (&ev.identity_a, &ev.identity_b) else { + return Verdict::NeedsReview("access control needs two identities — only one context was captured".into()); + }; + if a.identity == b.identity && !a.identity.is_empty() { + return Verdict::Rejected(format!("both requests used the same identity ('{}') — nothing crossed a boundary", a.identity)); + } + if b.status == 401 || b.status == 403 { + return Verdict::Rejected(format!("the other identity was denied ({}) — the control works", b.status)); + } + if b.status >= 400 { + return Verdict::Rejected(format!("the other identity got {} — no access was obtained", b.status)); + } + // A 200 that returns a login page or an empty shell is the classic + // false positive: the status says yes and the body says no. + if b.is_empty() { + return Verdict::NeedsReview("the other identity got 200 with an empty body — no resource content to compare".into()); + } + let overlap = semantic_overlap(&a.body, &b.body); + if overlap < 0.6 { + return Verdict::Rejected(format!( + "the other identity got 200 but the body does not match the owner's resource ({:.0}% overlap) — likely a login page or generic response", + overlap * 100.0 + )); + } + Verdict::Confirmed(format!( + "identity '{}' read identity '{}'s resource: {} with {:.0}% content match", + if b.identity.is_empty() { "B" } else { &b.identity }, + if a.identity.is_empty() { "A" } else { &a.identity }, + b.status, + overlap * 100.0 + )) + } +} + +pub struct SsrfValidator; +impl CweValidator for SsrfValidator { + fn name(&self) -> &'static str { + "ssrf" + } + fn cwes(&self) -> &'static [&'static str] { + &["918"] + } + fn evidence_required(&self) -> &'static [&'static str] { + &["controlled_callback OR private/canary resource retrieval"] + } + fn validate(&self, _f: &Finding, ev: &Evidence) -> Verdict { + if ev.callback_received && !ev.marker.is_empty() { + return Verdict::Confirmed(format!("out-of-band callback carrying marker {} was received", ev.marker)); + } + if ev.marker_observed && !ev.marker.is_empty() { + return Verdict::Confirmed(format!("the response returned content from the controlled internal resource ({})", ev.marker)); + } + let timing = ev + .baseline + .as_ref() + .zip(ev.attack.as_ref()) + .map(|(b, a)| diff(b, a).timing_delta_ms) + .unwrap_or(0); + if timing >= 4000 { + return Verdict::NeedsReview(format!("only a timing signal ({timing}ms) — consistent with SSRF but also with a slow upstream")); + } + Verdict::NeedsReview("no callback and no controlled resource retrieved — SSRF cannot be proven from the response alone".into()) + } +} + +pub struct LfiValidator; +impl CweValidator for LfiValidator { + fn name(&self) -> &'static str { + "lfi" + } + fn cwes(&self) -> &'static [&'static str] { + &["22", "23", "35", "98", "73"] + } + fn evidence_required(&self) -> &'static [&'static str] { + &["controlled_file_marker OR deterministic file content"] + } + fn validate(&self, _f: &Finding, ev: &Evidence) -> Verdict { + let Some(a) = &ev.attack else { + return Verdict::NeedsReview("no attack response captured".into()); + }; + if !ev.marker.is_empty() && a.body.contains(&ev.marker) { + return Verdict::Confirmed(format!("the response returned the controlled file marker {}", ev.marker)); + } + // Signatures of files that exist on essentially every host of that kind + // and cannot be produced by an application by accident. + const FILE_SIGS: &[(&str, &str)] = &[ + ("root:x:0:0", "/etc/passwd"), + ("daemon:x:1:1", "/etc/passwd"), + ("[boot loader]", "boot.ini"), + ("; for 16-bit app support", "win.ini"), + (" &'static str { + "rce" + } + fn cwes(&self) -> &'static [&'static str] { + &["77", "78", "94", "95", "502", "1336", "917"] + } + fn evidence_required(&self) -> &'static [&'static str] { + &["controlled side effect", "unique nonce", "output/callback confirmation"] + } + fn validate(&self, _f: &Finding, ev: &Evidence) -> Verdict { + if ev.marker.is_empty() { + return Verdict::NeedsReview("command execution needs a unique nonce the target could not produce on its own".into()); + } + let echoed = ev.attack.as_ref().map(|a| a.body.contains(&ev.marker)).unwrap_or(false); + if echoed && ev.marker_observed { + return Verdict::Confirmed(format!("the command's output carried the nonce {} back in the response", ev.marker)); + } + if ev.callback_received { + return Verdict::Confirmed(format!("the executed command called back with nonce {}", ev.marker)); + } + if echoed { + return Verdict::NeedsReview("the nonce appears in the response but was not confirmed as command output — it may just be reflected input".into()); + } + Verdict::NeedsReview("no nonce in the output and no callback — execution was not demonstrated".into()) + } +} + +/// Crude content-similarity: fraction of the owner's distinctive tokens that +/// also appear in the other identity's response. Enough to separate "the same +/// record" from "a login page with a 200 status", which is the distinction that +/// decides an IDOR. +fn semantic_overlap(a: &str, b: &str) -> f64 { + let toks: Vec<&str> = a + .split(|c: char| !c.is_alphanumeric()) + .filter(|t| t.len() >= 4) + .collect(); + if toks.is_empty() { + return 0.0; + } + let mut distinct: Vec<&str> = toks; + distinct.sort_unstable(); + distinct.dedup(); + let hits = distinct.iter().filter(|t| b.contains(**t)).count(); + hits as f64 / distinct.len() as f64 +} + +pub fn validators() -> Vec> { + vec![ + Box::new(SqliValidator), + Box::new(XssValidator), + Box::new(IdorValidator), + Box::new(SsrfValidator), + Box::new(LfiValidator), + Box::new(RceValidator), + ] +} + +/// The validator that owns this finding's class, if any. +pub fn validator_for(f: &Finding) -> Option> { + let n = cwe_num(&f.cwe); + if !n.is_empty() { + if let Some(v) = validators().into_iter().find(|v| v.cwes().contains(&n.as_str())) { + return Some(v); + } + } + // Fall back to the title when the agent omitted the CWE — the class is + // still the thing being claimed, and a missing field should not silently + // skip validation. + let t = f.title.to_lowercase(); + let by_title: &[(&str, fn() -> Box)] = &[ + ("sql injection", || Box::new(SqliValidator)), + ("sqli", || Box::new(SqliValidator)), + ("cross-site scripting", || Box::new(XssValidator)), + ("xss", || Box::new(XssValidator)), + ("idor", || Box::new(IdorValidator)), + ("broken access control", || Box::new(IdorValidator)), + ("bola", || Box::new(IdorValidator)), + ("ssrf", || Box::new(SsrfValidator)), + ("server-side request forgery", || Box::new(SsrfValidator)), + ("path traversal", || Box::new(LfiValidator)), + ("local file inclusion", || Box::new(LfiValidator)), + ("remote code execution", || Box::new(RceValidator)), + ("command injection", || Box::new(RceValidator)), + ]; + by_title.iter().find(|(k, _)| t.contains(k)).map(|(_, mk)| mk()) +} + +/// The final judge: the deterministic verdict, tempered by what the rest of the +/// pipeline already established. +/// +/// It can **downgrade** freely and **upgrade only within its own evidence**. A +/// class with no rule keeps whatever the vote decided but can never be silently +/// promoted to confirmed by this stage — the engine's job is to remove doubt it +/// can actually remove, not to add confidence it has not measured. +pub fn judge(f: &Finding, ev: Option<&Evidence>) -> Verdict { + let Some(v) = validator_for(f) else { + return Verdict::NeedsReview(format!( + "no deterministic validator for {} — kept for human review", + if f.cwe.is_empty() { "this class" } else { &f.cwe } + )); + }; + let Some(ev) = ev else { + return Verdict::NeedsReview(format!( + "{} requires {} — none was captured", + v.name(), + v.evidence_required().join(", ") + )); + }; + v.validate(f, ev) +} + +/// How forcefully the engine's verdict is applied. +/// +/// Agents have to *record* the artifacts before the engine can judge them, and +/// that contract is new. Turning enforcement on everywhere at once would mark +/// every finding from an agent that hasn't adopted it as unproven — technically +/// honest, operationally a regression. So the default is advisory: contradicted +/// findings are still rejected (that is a real measurement), but a finding the +/// votes confirmed is not demoted merely because no evidence was captured. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum Mode { + /// Engine disabled. + Off, + /// Record the verdict; reject contradictions; do not demote for absent evidence. + Advisory, + /// The verdict is the status. Nothing is confirmed without proof. + Enforcing, +} + +impl Mode { + /// `NEUROSPLOIT_VALIDATION=off|advisory|enforcing` (default advisory). + pub fn from_env() -> Mode { + match std::env::var("NEUROSPLOIT_VALIDATION").unwrap_or_default().trim().to_lowercase().as_str() { + "off" | "0" | "false" => Mode::Off, + "enforcing" | "enforce" | "strict" | "2" => Mode::Enforcing, + _ => Mode::Advisory, + } + } +} + +/// Judge a finding under `mode`, using the evidence the agent recorded on it. +pub fn apply_mode(f: &mut Finding, mode: Mode) -> Option { + if mode == Mode::Off { + return None; + } + let ev = f.evidence_data.clone(); + let verdict = judge(f, ev.as_ref()); + match (&verdict, mode) { + // A contradiction is a measurement, and it counts in either mode. + (Verdict::Rejected(_), _) => { + apply(f, ev.as_ref()); + } + (_, Mode::Enforcing) => { + apply(f, ev.as_ref()); + } + (Verdict::Confirmed(r), Mode::Advisory) => { + f.validated = true; + f.review_status = "confirmed".into(); + f.review_reason = format!("validated deterministically: {r}"); + f.confidence = f.confidence.max(0.9); + } + (Verdict::NeedsReview(r), Mode::Advisory) => { + // Leave the vote's verdict in place; say plainly that the + // deterministic engine could not corroborate it. + if f.review_reason.is_empty() { + f.review_reason = format!("not deterministically verified: {r}"); + } + } + (_, Mode::Off) => {} + } + Some(verdict) +} + +/// Apply the engine to a finding, updating `review_status`/`review_reason` and +/// `validated`. Returns the verdict for logging. +pub fn apply(f: &mut Finding, ev: Option<&Evidence>) -> Verdict { + let verdict = judge(f, ev); + match &verdict { + Verdict::Confirmed(r) => { + f.validated = true; + f.review_status = "confirmed".into(); + f.review_reason = format!("validated deterministically: {r}"); + f.confidence = f.confidence.max(0.9); + } + Verdict::NeedsReview(r) => { + // Never destroy a vote-confirmed finding on missing evidence — but + // never let it claim deterministic proof either. + f.validated = false; + f.review_status = "needs-review".into(); + f.review_reason = r.clone(); + f.confidence = f.confidence.min(0.7); + } + Verdict::Rejected(r) => { + f.validated = false; + f.review_status = "rejected".into(); + f.review_reason = format!("validator rejected: {r}"); + f.confidence = f.confidence.min(0.3); + } + } + verdict +} + +/// What the engine would need to confirm this class — rendered into exploit +/// prompts so agents collect the right artifacts *while* they have the target +/// in hand, instead of being asked for them after the run. +pub fn evidence_contract() -> String { + let mut s = String::from( + "EVIDENCE CONTRACT — a finding is only confirmed when the harness can verify it deterministically, without a model. Collect exactly this:\n", + ); + for v in validators() { + s.push_str(&format!(" {:<5} {}\n", v.name(), v.evidence_required().join(" · "))); + } + s.push_str(" Anything else is reported as needs-review. Record the baseline request, the attack request, and any marker the harness gave you.\n"); + s +} + +#[cfg(test)] +mod tests { + use super::*; + + fn ex(status: u16, body: &str) -> Exchange { + Exchange { method: "GET".into(), url: "https://t.test/x".into(), status, body: body.into(), ..Default::default() } + } + fn f(cwe: &str, title: &str) -> Finding { + Finding { cwe: cwe.into(), title: title.into(), confidence: 0.8, ..Default::default() } + } + + #[test] + fn sqli_needs_a_difference_that_repeats() { + let base = ex(200, "welcome user"); + let attack = ex(500, "You have an error in your SQL syntax near '1''"); + // One observation, no repeats: suspicious, not proven. + let once = Evidence { baseline: Some(base.clone()), attack: Some(attack.clone()), ..Default::default() }; + assert!(matches!(judge(&f("CWE-89", "SQLi"), Some(&once)), Verdict::NeedsReview(_))); + + let repeated = Evidence { + baseline: Some(base), + attack: Some(attack.clone()), + repeats: vec![attack.clone(), attack], + ..Default::default() + }; + match judge(&f("CWE-89", "SQLi"), Some(&repeated)) { + Verdict::Confirmed(r) => assert!(r.contains("sql syntax"), "{r}"), + v => panic!("expected confirmation, got {v:?}"), + } + } + + #[test] + fn sqli_is_rejected_when_the_payload_changed_nothing() { + let same = ex(200, "welcome user"); + let ev = Evidence { baseline: Some(same.clone()), attack: Some(same), ..Default::default() }; + assert!(matches!(judge(&f("CWE-89", "SQLi"), Some(&ev)), Verdict::Rejected(_))); + } + + #[test] + fn an_app_that_always_prints_sql_errors_does_not_count() { + let noisy = ex(200, "debug: sql syntax error somewhere"); + let ev = Evidence { + baseline: Some(noisy.clone()), + attack: Some(noisy.clone()), + repeats: vec![noisy.clone(), noisy], + ..Default::default() + }; + // Identical bodies: the error is not attributable to the payload. + assert!(matches!(judge(&f("CWE-89", "SQLi"), Some(&ev)), Verdict::Rejected(_))); + } + + #[test] + fn reflected_xss_without_a_browser_is_not_confirmed() { + let marker = canary("nsxss"); + let ev = Evidence { + marker: marker.clone(), + attack: Some(ex(200, &format!("
{marker}
"))), + ..Default::default() + }; + match judge(&f("CWE-79", "Reflected XSS"), Some(&ev)) { + Verdict::NeedsReview(r) => assert!(r.contains("no browser executed it"), "{r}"), + v => panic!("reflection alone must not confirm XSS: {v:?}"), + } + } + + #[test] + fn xss_is_confirmed_only_when_the_browser_reports_the_marker() { + let marker = canary("nsxss"); + let ev = Evidence { marker: marker.clone(), browser_executed: true, marker_observed: true, ..Default::default() }; + assert!(matches!(judge(&f("CWE-79", "XSS"), Some(&ev)), Verdict::Confirmed(_))); + } + + #[test] + fn idor_rejects_a_200_that_is_really_a_login_page() { + let owner = ex(200, "invoice 4711 total 1234.56 customer alice smith account 9981"); + let mut other = ex(200, "please sign in to continue"); + other.identity = "userB".into(); + let mut a = owner; + a.identity = "userA".into(); + let ev = Evidence { identity_a: Some(a), identity_b: Some(other), ..Default::default() }; + match judge(&f("CWE-639", "IDOR"), Some(&ev)) { + Verdict::Rejected(r) => assert!(r.contains("does not match"), "{r}"), + v => panic!("a login page with status 200 must not pass as IDOR: {v:?}"), + } + } + + #[test] + fn idor_confirms_when_the_other_identity_gets_the_owners_data() { + let body = "invoice 4711 total 1234.56 customer alice smith account 9981"; + let mut a = ex(200, body); + a.identity = "userA".into(); + let mut b = ex(200, body); + b.identity = "userB".into(); + let ev = Evidence { identity_a: Some(a), identity_b: Some(b), ..Default::default() }; + assert!(matches!(judge(&f("CWE-639", "IDOR"), Some(&ev)), Verdict::Confirmed(_))); + } + + #[test] + fn idor_rejects_when_the_control_actually_worked() { + let mut a = ex(200, "secret record"); + a.identity = "userA".into(); + let mut b = ex(403, "forbidden"); + b.identity = "userB".into(); + let ev = Evidence { identity_a: Some(a), identity_b: Some(b), ..Default::default() }; + match judge(&f("CWE-863", "BOLA"), Some(&ev)) { + Verdict::Rejected(r) => assert!(r.contains("denied"), "{r}"), + v => panic!("a 403 is the control working: {v:?}"), + } + } + + #[test] + fn ssrf_needs_a_callback_or_a_retrieved_resource() { + let ev = Evidence::default(); + assert!(matches!(judge(&f("CWE-918", "SSRF"), Some(&ev)), Verdict::NeedsReview(_))); + let ev2 = Evidence { marker: canary("nsoob"), callback_received: true, ..Default::default() }; + assert!(matches!(judge(&f("CWE-918", "SSRF"), Some(&ev2)), Verdict::Confirmed(_))); + } + + #[test] + fn lfi_confirms_on_a_file_signature_the_baseline_lacked() { + let ev = Evidence { + baseline: Some(ex(200, "normal page")), + attack: Some(ex(200, "root:x:0:0:root:/root:/bin/bash\ndaemon:x:1:1:")), + ..Default::default() + }; + match judge(&f("CWE-22", "Path traversal"), Some(&ev)) { + Verdict::Confirmed(r) => assert!(r.contains("/etc/passwd"), "{r}"), + v => panic!("expected confirmation: {v:?}"), + } + } + + #[test] + fn rce_rejects_a_nonce_that_is_only_reflected_input() { + let nonce = canary("nsrce"); + let ev = Evidence { marker: nonce.clone(), attack: Some(ex(200, &format!("you searched for {nonce}"))), ..Default::default() }; + match judge(&f("CWE-78", "Command injection"), Some(&ev)) { + Verdict::NeedsReview(r) => assert!(r.contains("reflected input"), "{r}"), + v => panic!("reflection is not execution: {v:?}"), + } + } + + #[test] + fn a_class_without_a_validator_is_never_auto_confirmed() { + let v = judge(&f("CWE-1004", "Cookie without HttpOnly"), None); + assert!(matches!(v, Verdict::NeedsReview(_)), "got {v:?}"); + } + + #[test] + fn missing_evidence_is_review_not_confirmation() { + let v = judge(&f("CWE-89", "SQL Injection in id"), None); + match v { + Verdict::NeedsReview(r) => assert!(r.contains("none was captured"), "{r}"), + v => panic!("absent evidence must never confirm: {v:?}"), + } + } + + #[test] + fn the_title_routes_the_finding_when_the_cwe_is_missing() { + let mut finding = f("", "Reflected Cross-Site Scripting in search"); + let marker = canary("nsxss"); + let ev = Evidence { marker, browser_executed: true, marker_observed: true, ..Default::default() }; + let verdict = apply(&mut finding, Some(&ev)); + assert!(matches!(verdict, Verdict::Confirmed(_))); + assert_eq!(finding.review_status, "confirmed"); + assert!(finding.validated); + } + + #[test] + fn apply_downgrades_confidence_when_it_cannot_prove_the_claim() { + let mut finding = Finding { cwe: "CWE-89".into(), title: "SQLi".into(), confidence: 0.95, validated: true, ..Default::default() }; + apply(&mut finding, None); + assert!(!finding.validated); + assert_eq!(finding.review_status, "needs-review"); + assert!(finding.confidence <= 0.7, "confidence must not survive unproven: {}", finding.confidence); + } + + #[test] + fn advisory_mode_keeps_a_voted_finding_but_says_it_is_unproven() { + let mut finding = Finding { cwe: "CWE-89".into(), title: "SQLi".into(), confidence: 0.9, validated: true, review_status: "confirmed".into(), ..Default::default() }; + apply_mode(&mut finding, Mode::Advisory); + assert!(finding.validated, "advisory must not demote on absent evidence"); + assert!(finding.review_reason.contains("not deterministically verified"), "{}", finding.review_reason); + } + + #[test] + fn enforcing_mode_demotes_the_same_finding() { + let mut finding = Finding { cwe: "CWE-89".into(), title: "SQLi".into(), confidence: 0.9, validated: true, review_status: "confirmed".into(), ..Default::default() }; + apply_mode(&mut finding, Mode::Enforcing); + assert!(!finding.validated); + assert_eq!(finding.review_status, "needs-review"); + } + + #[test] + fn a_contradiction_is_rejected_even_in_advisory_mode() { + let same = ex(200, "welcome user"); + let mut finding = Finding { + cwe: "CWE-89".into(), + title: "SQLi".into(), + confidence: 0.9, + validated: true, + evidence_data: Some(Evidence { baseline: Some(same.clone()), attack: Some(same), ..Default::default() }), + ..Default::default() + }; + apply_mode(&mut finding, Mode::Advisory); + assert_eq!(finding.review_status, "rejected"); + } + + #[test] + fn canaries_do_not_repeat() { + // The regression: minting two in the same clock tick returned the same + // token, which would let a stale marker vouch for a new finding. + let batch: Vec = (0..500).map(|_| canary("ns")).collect(); + let mut uniq = batch.clone(); + uniq.sort(); + uniq.dedup(); + assert_eq!(uniq.len(), batch.len(), "canaries must be unique even when minted back to back"); + assert!(batch[0].starts_with("ns") && batch[0].len() > 8); + } +}