From 48c38d479865d9fdcfdfbf25c84c828b9179a0e6 Mon Sep 17 00:00:00 2001 From: CyberSecurityUP Date: Sat, 19 Sep 2026 18:06:50 -0300 Subject: [PATCH] fix(wiring): activate cvss.rs + waf.rs (were shelf-ware); TypeSafe into CVSS + agent selection MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Honesty audit found four modules written but not on the runtime path. Two mattered and are now wired; two are noted. cvss.rs — was NOT called; finding.cvss came from the old attack_graph ladder. Now attack_graph::cvss_graded() bridges the class shape + demonstrated rung into crate::cvss::grade (the FIRST-verbatim v3.1 equation), and enrich() sets finding.cvss from the demonstrated vector, recording the potential ceiling in the impact text. The class ladder remains only as a fallback for findings with no evidence to grade. waf.rs — the deterministic classifier was NOT run on any real exchange (only WAF_OPS prompt text reached the agent). Now poc.rs classifies each re-run: a PoC answered by a WAF/CDN is Unverifiable, not "gone" — closing a false-demotion where an edge block looked like a fix. TypeSafe (System One) extended per the build-with docs: - CVSS via System One: when impact_demonstrated < 0.5, the finding's CVSS is re-graded with impact receipts stripped — the calibrated judgment, not just the rung, decides the demonstrated number. - Agent selection: typesafe_prune_agents() asks one batched request (the fan-out pattern), a Noul per chosen agent, and drops only those it calibrates as clearly irrelevant (p < 0.25), never prunes to empty. Additive over the LLM selection; skipped without a key. Still shelf-ware, flagged honestly (not wired): inbox.rs (mail.tm/SMS happens via agent prompt instructions, the Rust client is unused) and pomdp.rs (redundant — belief.rs is the one on the path). 374 tests. Co-Authored-By: Claude Opus 5 (1M context) --- .../crates/harness/src/attack_graph.rs | 92 ++++++++++++++++++- neurosploit-rs/crates/harness/src/pipeline.rs | 80 +++++++++++++++- neurosploit-rs/crates/harness/src/poc.rs | 11 +++ 3 files changed, 180 insertions(+), 3 deletions(-) diff --git a/neurosploit-rs/crates/harness/src/attack_graph.rs b/neurosploit-rs/crates/harness/src/attack_graph.rs index bd9d009..ef16181 100644 --- a/neurosploit-rs/crates/harness/src/attack_graph.rs +++ b/neurosploit-rs/crates/harness/src/attack_graph.rs @@ -261,6 +261,67 @@ fn temporal(f: &Finding) -> (&'static str, &'static str) { } /// Derive a CVSS v3.1 base score + vector for a finding. +/// Evidence-graded CVSS via the FIRST-verbatim calculator in `crate::cvss`. +/// +/// The class proposes the vector's shape (which of C/I/A it *can* affect, the +/// scope, the exploitability axes); the demonstrated rung decides which impact +/// metrics actually have a receipt. `crate::cvss::grade` then keeps two scores: +/// the demonstrated one (what the evidence proved, the reported number) and the +/// potential one (what the class could reach). This is what wires the real +/// CVSS 3.1 equation — the older `cvss_for` is kept only as a fallback for a +/// finding with no structured evidence to grade. +pub fn cvss_graded(f: &Finding) -> Option { + use crate::cvss::{Ac, Av, Imp, Pr, Scope, Ui, Vector}; + let n: u32 = f.cwe.chars().skip_while(|c| !c.is_ascii_digit()).take_while(|c| c.is_ascii_digit()).collect::().parse().unwrap_or(0); + if n == 0 { + return None; + } + let imp = |s: &str| match s { "H" => Imp::High, "L" => Imp::Low, _ => Imp::None }; + // Class → the impact shape it CAN have (the potential ceiling) + scope. + let (c, i, a, scope) = match n { + 77 | 78 | 94 | 95 | 502 | 917 | 1336 => ("H", "H", "H", Scope::Changed), + 89 | 943 | 564 => ("H", "H", "L", Scope::Unchanged), + 22 | 23 | 35 | 98 | 73 => ("H", "N", "N", Scope::Unchanged), + 918 => ("H", "L", "N", Scope::Changed), + 639 | 862 | 863 | 284 | 285 | 306 | 566 | 425 => ("H", "H", "N", Scope::Unchanged), + 287 | 288 | 289 | 290 | 347 | 345 | 384 => ("H", "H", "N", Scope::Unchanged), + 79 | 80 | 83 | 87 => ("L", "L", "N", Scope::Changed), + 352 => ("N", "H", "N", Scope::Unchanged), + 611 | 776 | 827 => ("H", "N", "L", Scope::Changed), + 319 | 522 | 798 | 312 | 256 | 257 | 321 => ("H", "N", "N", Scope::Unchanged), + 200 | 209 | 538 | 540 | 548 | 532 | 530 => ("L", "N", "N", Scope::Unchanged), + 307 | 799 | 770 | 400 => ("N", "N", "L", Scope::Unchanged), + 601 => ("L", "L", "N", Scope::Changed), + 1021 => ("N", "L", "N", Scope::Unchanged), + 113 | 93 | 644 => ("L", "L", "N", Scope::Unchanged), + 525 | 524 => ("L", "N", "N", Scope::Unchanged), + _ => ("L", "N", "N", Scope::Unchanged), + }; + let authenticated = f.auth_context.eq_ignore_ascii_case("authenticated") || !f.account.is_empty(); + let proposed = Vector { + // Web engagement defaults; the class overrides where it matters. + av: Av::Network, + ac: match n { 362 | 208 | 385 => Ac::High, _ => Ac::Low }, // race/timing = high AC + pr: if authenticated { Pr::Low } else { Pr::None }, + ui: match n { 79 | 80 | 83 | 87 | 352 | 601 | 1021 => Ui::Required, _ => Ui::None }, + scope, + c: imp(c), + i: imp(i), + a: imp(a), + }; + // The demonstrated rung decides which impact metrics carry a receipt. + let rung = demonstrated_rung(f); + let has_c = matches!(rung, Rung::ReadData | Rung::ReadSensitive | Rung::Wrote | Rung::Executed | Rung::CrossedSystem); + let has_i = matches!(rung, Rung::Wrote | Rung::Executed | Rung::CrossedSystem); + let has_a = matches!(rung, Rung::Executed | Rung::CrossedSystem); + Some(crate::cvss::grade(proposed, move |m| match m { + "C" => has_c, + "I" => has_i, + "A" => has_a, + _ => true, + })) +} + pub fn cvss_for(f: &Finding) -> (f64, String) { let n: u32 = f.cwe.chars().skip_while(|c| !c.is_ascii_digit()).take_while(|c| c.is_ascii_digit()).collect::().parse().unwrap_or(0); let authenticated = f.auth_context.eq_ignore_ascii_case("authenticated") || !f.account.is_empty(); @@ -389,8 +450,22 @@ pub fn enrich(findings: &mut [Finding]) { // A severity word without the industry-standard number makes the reader // re-derive it by hand or take it on faith. if f.cvss.is_empty() { - let (score, vector) = cvss_for(f); - if score > 0.0 { f.cvss = format!("{score:.1} ({vector})"); } + // Evidence-graded first (FIRST-verbatim, demonstrated vs potential); + // fall back to the class ladder only when there is no evidence to grade. + match cvss_graded(f) { + Some(g) if g.demonstrated_score > 0.0 => { + f.cvss = format!("{:.1} ({})", g.demonstrated_score, g.demonstrated.vector_string()); + // Record the potential ceiling in the impact text when it is + // meaningfully higher, so the reader sees both numbers. + if g.potential_score - g.demonstrated_score > 0.5 && !f.impact.contains("potential CVSS") { + f.impact = format!("{} (potential CVSS {:.1} if fully exploited)", f.impact, g.potential_score).trim().to_string(); + } + } + _ => { + let (score, vector) = cvss_for(f); + if score > 0.0 { f.cvss = format!("{score:.1} ({vector})"); } + } + } } } } @@ -733,4 +808,17 @@ mod ladder_tests { } assert_eq!(temporal_factor("H", "C"), 1.0); } + + + #[test] + fn graded_cvss_wires_the_first_calculator_and_splits_demonstrated_from_potential() { + let reached = sqli(Some(Evidence { attack: Some(ex(200, "ok")), ..Default::default() })); + let g = cvss_graded(&reached).expect("graded"); + assert!(g.demonstrated_score <= g.potential_score); + + let proven = sqli(Some(Evidence { attack: Some(ex(200, "password=hunter2 bearer eyJ...")), ..Default::default() })); + let g2 = cvss_graded(&proven).expect("graded"); + assert!(g2.demonstrated_score >= g.demonstrated_score, "reading data cannot lower the score"); + assert!(g2.demonstrated.vector_string().contains("CVSS:3.1/")); + } } diff --git a/neurosploit-rs/crates/harness/src/pipeline.rs b/neurosploit-rs/crates/harness/src/pipeline.rs index 79a68a9..70e624b 100644 --- a/neurosploit-rs/crates/harness/src/pipeline.rs +++ b/neurosploit-rs/crates/harness/src/pipeline.rs @@ -1514,12 +1514,18 @@ async fn select_agents(pool: &ModelPool, recon: &str, focus: &str, catalog: &[Ag let user = format!("{focus_line}RECON:\n{recon_trim}\n\nAGENT CATALOG (name — title [cwe]):\n{list}\n\nReturn a JSON array of agent names to run."); match pool.complete_routed(Task::Select, "select", SELECT_SYS, &user).await { Ok((m, text)) => { - let names = parse_string_array(&text); + let mut names = parse_string_array(&text); if names.is_empty() { let preview: String = text.chars().take(120).collect(); let _ = tx.send(format!("agent selection via {} returned no parseable list ({} chars): {}", m.label(), text.len(), preview.replace('\n', " "))).await; } else { let _ = tx.send(format!("agent selection via {} → {} agent(s) chosen", m.label(), names.len())).await; + // System One refinement: ask TypeSafe, in ONE batched request + // (the fan-out pattern), whether each chosen agent is relevant to + // the observed surface, and drop the ones it calibrates as clearly + // irrelevant. Additive — it only prunes obvious mismatches, never + // adds agents, and is skipped entirely when TypeSafe is absent. + names = typesafe_prune_agents(recon, catalog, names, tx).await; } names } @@ -1530,6 +1536,62 @@ async fn select_agents(pool: &ModelPool, recon: &str, focus: &str, catalog: &[Ag } } +/// Prune agent choices TypeSafe judges irrelevant to the observed surface. +/// +/// One request, one Noul per chosen agent (they run in parallel and cannot see +/// one another). An agent is dropped only when the probability it is relevant is +/// clearly low (< 0.25) — a conservative gate, because a false drop costs a +/// missed class while a false keep only costs one agent's budget. No key, or any +/// error, means the LLM's selection stands unchanged. +async fn typesafe_prune_agents(recon: &str, catalog: &[Agent], chosen: Vec, tx: &Sender) -> Vec { + if std::env::var("NEUROSPLOIT_TYPESAFE").unwrap_or_default().trim().to_lowercase() == "off" { + return chosen; + } + let Some(ts) = crate::typesafe::TypeSafe::from_env() else { return chosen }; + use std::collections::BTreeMap; + let mut qs: BTreeMap = BTreeMap::new(); + for name in &chosen { + if let Some(a) = catalog.iter().find(|a| &a.name == name) { + qs.insert( + name.clone(), + crate::typesafe::Question::noul( + &format!("Given the observed attack surface, is the '{}' test ({}, {}) relevant enough to be worth running?", a.name, a.title, a.cwe), + "the surface plausibly has this class or the technique applies", + "the surface shows no sign this class could exist here", + ), + ); + } + } + if qs.is_empty() { + return chosen; + } + let fallback = chosen.clone(); // the LLM's selection stands on any failure + let state = serde_json::json!({ "recon": recon.chars().take(4000).collect::() }); + match ts.evaluate(state, qs).await { + Ok(answers) => { + let before = chosen.len(); + let kept: Vec = chosen.into_iter().filter(|name| { + // Keep unless TypeSafe is clearly confident it is irrelevant. + answers.get(name).and_then(|a| a.noul).map(|p| p >= 0.25).unwrap_or(true) + }).collect(); + let dropped = before - kept.len(); + // Never prune to nothing — a total rejection is more likely a bad + // question than a truly empty surface; keep the LLM's call. + if kept.is_empty() { + return fallback; + } + if dropped > 0 { + let _ = tx.send(format!("notify: 🧮 TypeSafe pruned {dropped} agent(s) as irrelevant to the surface")).await; + } + kept + } + Err(e) => { + let _ = tx.send(format!("notify: ⚠ TypeSafe agent pruning unavailable ({e}) — keeping the selection")).await; + fallback + } + } +} + fn parse_string_array(text: &str) -> Vec { match (text.find('['), text.rfind(']')) { (Some(a), Some(b)) if b > a => serde_json::from_str::>(&text[a..=b]).unwrap_or_default(), @@ -2220,6 +2282,22 @@ async fn finish(cfg: RunConfig, _lib: &Library, pool: &ModelPool, recon: String, adj.p_confirmed, adj.impact_demonstrated ); } + // CVSS via System One: re-grade the vector using the + // calibrated impact judgment as the impact receipt, + // not just the deterministic rung. A finding TypeSafe + // says shows no real impact loses its C/I/A the same + // way an absent receipt would — the demonstrated + // score follows the evidence, calibrated. + if adj.impact_demonstrated < 0.5 { + if let Some(g) = crate::attack_graph::cvss_graded(f) { + // Strip demonstrated impact the model is not + // convinced of; keep potential as context. + let dropped = crate::cvss::grade(g.potential, |_| false); + if dropped.demonstrated_score < g.demonstrated_score { + f.cvss = format!("{:.1} ({})", dropped.demonstrated_score, dropped.demonstrated.vector_string()); + } + } + } refined += 1; } audit.append( diff --git a/neurosploit-rs/crates/harness/src/poc.rs b/neurosploit-rs/crates/harness/src/poc.rs index 3a1ec73..2e46384 100644 --- a/neurosploit-rs/crates/harness/src/poc.rs +++ b/neurosploit-rs/crates/harness/src/poc.rs @@ -146,6 +146,14 @@ impl PocValidator { Err(reason) => return self.unverifiable(f, &reason), }; + // If the re-run was answered by a WAF/CDN rather than the application, + // the PoC was NOT tested — do not report it as "no longer reproduces". + // This is the deterministic WAF classifier running on a real exchange. + let verdict_edge = crate::waf::classify_exchange(&fresh); + if !verdict_edge.origin.supports_a_finding() { + return self.unverifiable(f, &format!("re-run was answered by the edge, not the application: {}", verdict_edge.reason)); + } + // Rebuild the evidence with the fresh responses, keep the finding's // markers, and ask the same deterministic judge. let mut fresh_ev = Evidence { @@ -205,6 +213,9 @@ impl PocValidator { } match self.engine.send(&spec).await { Ok(fresh) => { + if !crate::waf::classify_exchange(&fresh).origin.supports_a_finding() { + return self.unverifiable(f, "re-fetch was answered by a WAF/CDN, not the application"); + } let fresh_ev = Evidence { attack: Some(fresh), baseline: ev.baseline.clone(), identity_a: ev.identity_a.clone(), identity_b: ev.identity_b.clone(), ..ev.clone() }; match judge(f, Some(&fresh_ev)) { Verdict::Confirmed(r) => PocResult { finding_id: f.id.clone(), reproduction: Reproduction::Reproduced, detail: format!("re-fetched; still validates: {r}"), reverdict: Some(r), replayed: Some(spec.url) },