diff --git a/BENCHMARK.md b/BENCHMARK.md index a4acf0a..4ad1e25 100644 --- a/BENCHMARK.md +++ b/BENCHMARK.md @@ -33,6 +33,9 @@ The tools compared: [Strix](https://github.com/usestrix/strix) (Apache 2.0), | Hash-chained audit trail | — | — | — | ✅ | | OT/SCADA/ICS safety policy | — | — | — | ✅ | | Internal network / AD attack graph | — | — | — | ✅ | +| Self-hosted OOB channel (blind SSRF/XXE/RCE) | via tools | — | ✅ Burp | ✅ own DNS+HTTP listeners | +| Fail-closed egress (VPN/bastion/tunnel) | — | — | — | ✅ | +| WAF-aware inference (block ≠ "not vulnerable") | — | — | — | ✅ | | FAIR loss quantification | — | — | — | ✅ | | Provenance / watermarking | — | — | — | ✅ | | Published benchmark results | dir exists, empty | — | marketing | ❌ **none, including this one** | @@ -112,9 +115,9 @@ attacker-supplied-shaped payloads this is the largest single gap in the comparison, and the next thing worth building. **2. No real intercepting proxy.** Strix ships Caido integration; Penligent -drives Burp. NeuroSploit can route through an upstream proxy but does not own -the request/response stream, which limits replay fidelity and passive -discovery. +drives Burp. NeuroSploit can route through an upstream proxy — and now through +a VPN, bastion, or Cloudflare tunnel, fail-closed — but it does not own the +request/response stream, which limits replay fidelity and passive discovery. **3. Nobody has run it against a benchmark.** Strix has an empty `benchmarks/` directory, Shannon publishes none, and neither does this project. Until @@ -164,9 +167,9 @@ a comparison of intentions. |---|---| | Agents / skills | 446 (255 vulnerability, plus recon, code, infra, AI, chains, meta) | | Deterministic validators | 22 CWE classes with evidence preconditions | -| Rust modules | 32 | +| Rust modules | 37 | | Rust LOC | ~24k | -| Tests | 260, all passing | +| Tests | 296, all passing | ## Next, to make this a real benchmark diff --git a/neurosploit-rs/crates/harness/src/lib.rs b/neurosploit-rs/crates/harness/src/lib.rs index a907e48..060f7d2 100644 --- a/neurosploit-rs/crates/harness/src/lib.rs +++ b/neurosploit-rs/crates/harness/src/lib.rs @@ -40,6 +40,7 @@ pub mod transport; pub mod types; pub mod uncertainty; pub mod validation; +pub mod waf; pub use agents::{Agent, Library}; pub use models::{ diff --git a/neurosploit-rs/crates/harness/src/pipeline.rs b/neurosploit-rs/crates/harness/src/pipeline.rs index f574878..df72175 100644 --- a/neurosploit-rs/crates/harness/src/pipeline.rs +++ b/neurosploit-rs/crates/harness/src/pipeline.rs @@ -371,6 +371,7 @@ fn engagement_ops(cfg: &RunConfig) -> String { }; let oob = oob_ops(cfg); let sms = sms_ops(cfg); + let waf = WAF_OPS; format!( "ENGAGEMENT OPS — TEST ACCOUNTS & VAULT:\n\ - CREDENTIAL VAULT: whenever you create a test account or generate any credential, APPEND one JSON line to \ @@ -385,7 +386,7 @@ fn engagement_ops(cfg: &RunConfig) -> String { explicit about which findings needed a login. In black-box, record in `how`/evidence exactly what you did \ to create the user.\n\ - {temp}\n\ - {oob}{sms}\n" + {oob}{sms}{waf}\n" ) } @@ -408,6 +409,13 @@ fn oob_ops(cfg: &RunConfig) -> String { ) } +/// What an agent must do when the edge answers instead of the application. +/// +/// Both failure directions are named explicitly, because agents make both: a +/// 403 from a WAF read as "not vulnerable" (the expensive one), and a block +/// page echoing the payload read as reflection (the embarrassing one). +const WAF_OPS: &str = "- WAF / EDGE: if a response came from a CDN or WAF rather than the application (vendor headers plus block-page wording, a challenge, or HTTP 429), the application NEVER SAW your request. Never record that as 'tested, not vulnerable' — record that the control could not be reached, and say which probes were blocked. A payload echoed back by a block page is the EDGE reflecting it, not the application: it is not XSS evidence. On a 429 or a browser challenge, pace the requests and retry — that is not a verdict on the payload.\n"; + /// Inbound SMS instructions, when a number is configured. fn sms_ops(cfg: &RunConfig) -> String { match cfg.sms.as_deref().filter(|s| !s.trim().is_empty()) { diff --git a/neurosploit-rs/crates/harness/src/waf.rs b/neurosploit-rs/crates/harness/src/waf.rs new file mode 100644 index 0000000..e93b2c2 --- /dev/null +++ b/neurosploit-rs/crates/harness/src/waf.rs @@ -0,0 +1,458 @@ +//! WAF awareness — telling the edge apart from the application. +//! +//! When a WAF sits in front of a target, every response an agent reads may +//! have been written by the edge rather than by the application. That breaks +//! inference in two opposite directions at once, and both are common: +//! +//! ```text +//! payload → 403 from Cloudflare → "not vulnerable" ← false NEGATIVE +//! payload → block page echoing it → "payload reflected!" ← false POSITIVE +//! ``` +//! +//! The first is the expensive one. A WAF blocking a probe says nothing about +//! the code behind it: the application may be wide open and simply never +//! reached. Reporting "SQL injection tested, not vulnerable" on the strength of +//! a 403 from the edge is a statement about the WAF, not the app — and it is +//! the statement that gets a real bug missed. +//! +//! The second is the embarrassing one. A block page frequently includes the +//! offending payload ("Your request contained: `