diff --git a/README.md b/README.md index 944bc4b..b111ea3 100755 --- a/README.md +++ b/README.md @@ -1,4 +1,4 @@ -

๐Ÿง  NeuroSploit v4.2.0

+

๐Ÿง  NeuroSploit v4.2.1

Stars @@ -8,7 +8,7 @@

- + @@ -52,6 +52,15 @@ Control TUI**. ### Highlights +> **New in v4.2.1** โ€” **SARIF 2.1.0 export**: every run now writes `report.sarif` +> next to the Markdown/JSON/HTML/PDF, and `neurosploit sarif ` (re)emits it +> on demand, so findings drop straight into GitHub / Azure DevOps code-scanning +> as severity-coloured, CWE-linked alerts (also exposed over MCP). Plus stronger +> **cross-object reference mining** in the chaining loop โ€” the engine harvests +> every object identifier it sees (ids, UUIDs, tokens, emails) into a reference +> pool and substitutes them across identities and endpoints, the core of +> reliable BOLA / IDOR / mass-assignment discovery. + > **New in v4.2.0** โ€” **binary / APK / IPA testing**: a new `mobile` mode analyses > a local artifact with 12 reverse-engineering skills (static binary triage, > APK/IPA static analysis, RASP & anti-tamper mapping, root/jailbreak, TLS @@ -741,6 +750,13 @@ a `.http` file: neurosploit traffic # flows.jsonl -> traffic.http ``` +Every run also writes `report.sarif` (SARIF 2.1.0); re-emit it any time for CI +code-scanning: + +```bash +neurosploit sarif # findings -> report.sarif (GitHub/Azure code-scanning) +``` + --- ## ๐Ÿ”Œ Run it as an MCP server @@ -761,8 +777,8 @@ and authorization are identical to the CLI. See TUTORIAL section 8. ## ๐Ÿ“Š How we compare -A rough, honest capability benchmark against Strix, Shannon, Penligent and the -other open-source agents โ€” including where NeuroSploit is **behind** (no +A rough, honest capability benchmark against Shannon, Penligent and other +open-source agents โ€” including where NeuroSploit is **behind** (no container isolation, no real intercepting proxy, no published benchmark run) โ€” lives in **[BENCHMARK.md](BENCHMARK.md)**. @@ -958,6 +974,7 @@ Every run writes a self-contained folder `runs/ns--/`: | `exploitation.md` | raw per-agent transcript | | `findings.json` / `findings.md` | validated findings (reuse by other tools/AIs) | | `report.html`, `report.typ`, `report.pdf` | final report (PDF via the Typst engine) | +| `report.sarif` | SARIF 2.1.0 results for CI code-scanning ingestion | A reinforcement-learning reward store (`data/rl_state_rs.json`) biases agent selection on future runs. diff --git a/neurosploit-rs/Cargo.lock b/neurosploit-rs/Cargo.lock index ff71b19..a3f6100 100644 --- a/neurosploit-rs/Cargo.lock +++ b/neurosploit-rs/Cargo.lock @@ -929,7 +929,7 @@ dependencies = [ [[package]] name = "neurosploit" -version = "4.2.0" +version = "4.2.1" dependencies = [ "anyhow", "clap", @@ -946,7 +946,7 @@ dependencies = [ [[package]] name = "neurosploit-harness" -version = "4.2.0" +version = "4.2.1" dependencies = [ "anyhow", "base64", diff --git a/neurosploit-rs/Cargo.toml b/neurosploit-rs/Cargo.toml index 3fec3ff..6e08880 100644 --- a/neurosploit-rs/Cargo.toml +++ b/neurosploit-rs/Cargo.toml @@ -3,7 +3,7 @@ members = ["crates/harness", "app"] resolver = "2" [workspace.package] -version = "4.2.0" +version = "4.2.1" edition = "2021" license = "MIT" repository = "https://github.com/JoasASantos/NeuroSploit" diff --git a/neurosploit-rs/app/src/main.rs b/neurosploit-rs/app/src/main.rs index af8cd5a..da9e6f5 100644 --- a/neurosploit-rs/app/src/main.rs +++ b/neurosploit-rs/app/src/main.rs @@ -203,6 +203,15 @@ enum Cmd { /// Run id or path. run: String, }, + /// Emit SARIF 2.1.0 for a finished run so CI code-scanning (GitHub, Azure + /// DevOps) can ingest the findings as annotated, severity-coloured alerts. + Sarif { + /// Run id (`ns-โ€ฆ`) or a path to the run directory. + run: String, + /// Write to this path instead of the run's `report.sarif`. + #[arg(long = "out")] + out: Option, + }, /// Verify a finished run's audit trail โ€” the hash chain and, with --anchor, /// the signed anchors that catch truncation and silent rebuilds. Audit { @@ -808,6 +817,7 @@ async fn main() -> anyhow::Result<()> { } Cmd::Audit { run, anchor } => handle_audit(&base, &run, anchor)?, Cmd::Traffic { run } => handle_traffic(&base, &run)?, + Cmd::Sarif { run, out } => handle_sarif(&base, &run, out.as_deref())?, Cmd::Assurance { run, verify } => handle_assurance(&base, &run, verify)?, Cmd::Compliance { run, framework, include_leads } => handle_compliance(&base, &run, &framework, include_leads)?, Cmd::Poc { run, repeats, apply } => handle_poc(&base, &run, repeats, apply).await?, @@ -1657,6 +1667,21 @@ fn handle_traffic(base: &std::path::Path, run: &str) -> anyhow::Result<()> { Ok(()) } +fn handle_sarif(base: &std::path::Path, run: &str, out: Option<&str>) -> anyhow::Result<()> { + let dir = resolve_run(base, run)?; + let findings = load_findings(&dir)?; + let target = std::fs::read_to_string(dir.join("meta.json")) + .ok() + .and_then(|t| serde_json::from_str::(&t).ok()) + .and_then(|v| v.get("target").and_then(|x| x.as_str()).map(String::from)) + .unwrap_or_else(|| run.to_string()); + let doc = harness::sarif::to_string(&target, &findings); + let dest = out.map(std::path::PathBuf::from).unwrap_or_else(|| dir.join("report.sarif")); + std::fs::write(&dest, doc)?; + println!(" {} finding(s) -> SARIF 2.1.0 at {}", findings.len(), dest.display()); + Ok(()) +} + fn handle_audit(base: &std::path::Path, run: &str, anchor: bool) -> anyhow::Result<()> { let dir = resolve_run(base, run)?; let log = harness::audit::AuditLog::open(dir.join("audit.jsonl")); diff --git a/neurosploit-rs/app/src/mcp.rs b/neurosploit-rs/app/src/mcp.rs index c71ce0d..70c8d5b 100644 --- a/neurosploit-rs/app/src/mcp.rs +++ b/neurosploit-rs/app/src/mcp.rs @@ -99,6 +99,7 @@ fn tool_list() -> Value { { "name": "neurosploit_findings", "description": "Read a finished run's findings as JSON.", "inputSchema": { "type": "object", "properties": { "run": { "type": "string", "description": "Run id or path" } }, "required": ["run"] } }, { "name": "neurosploit_report", "description": "Read a finished run's Markdown report.", "inputSchema": { "type": "object", "properties": { "run": { "type": "string" } }, "required": ["run"] } }, { "name": "neurosploit_rebuild", "description": "Rebuild a run's report artifacts from its findings (no model calls).", "inputSchema": { "type": "object", "properties": { "run": { "type": "string" } }, "required": ["run"] } }, + { "name": "neurosploit_sarif", "description": "Emit SARIF 2.1.0 for a finished run (report.sarif) so CI code-scanning can ingest the findings.", "inputSchema": { "type": "object", "properties": { "run": { "type": "string" } }, "required": ["run"] } }, { "name": "neurosploit_internal", "description": "Internal-network / Active Directory attack-graph analysis: paths to crown jewels and the choke point to fix first.", "inputSchema": { "type": "object", "properties": { "graph": { "type": "string", "description": "Path to a graph JSON" }, "scaffold": { "type": "string", "description": "Domain to scaffold, e.g. corp.local" }, "from": { "type": "string", "description": "Foothold node id" } } } }, { "name": "neurosploit_compliance", "description": "Map a finished run's findings onto PCI-DSS, HIPAA or SOC 2 controls.", "inputSchema": { "type": "object", "properties": { "run": { "type": "string" }, "framework": { "type": "string", "enum": ["pci-dss","hipaa","soc2"] } }, "required": ["run"] } }, { "name": "neurosploit_container", "description": "Scan an OCI container image (repo:tag / tar / Dockerfile) for vulnerable packages, secrets, misconfig and emit an SBOM.", "inputSchema": { "type": "object", "properties": { "image": { "type": "string" }, "model": { "type": "string" }, "subscription": { "type": "boolean" } }, "required": ["image"] } } @@ -138,6 +139,7 @@ fn handle_call(id: Option, req: &Value, exe: &std::path::Path) -> Value { return read_run_file(id, &run, "report.md"); } "neurosploit_rebuild" => { let Some(run) = s("run") else { return tool_err(id, "run is required") }; argv.push("rebuild".into()); argv.push(run); } + "neurosploit_sarif" => { let Some(run) = s("run") else { return tool_err(id, "run is required") }; argv.push("sarif".into()); argv.push(run); } "neurosploit_internal" => { argv.push("internal".into()); if let Some(g) = s("graph") { argv.push("--graph".into()); argv.push(g); } diff --git a/neurosploit-rs/crates/harness/src/assurance.rs b/neurosploit-rs/crates/harness/src/assurance.rs index 2208add..daa0cac 100644 --- a/neurosploit-rs/crates/harness/src/assurance.rs +++ b/neurosploit-rs/crates/harness/src/assurance.rs @@ -86,6 +86,7 @@ const KNOWN: &[(&str, &str)] = &[ ("flows.jsonl", "intercepted request/response flows"), ("meta.json", "target metadata"), ("coverage.md", "what was tested and what was not"), + ("report.sarif", "SARIF 2.1.0 results for CI code-scanning ingestion"), ]; fn hash_file(path: &Path) -> Option<(String, u64)> { diff --git a/neurosploit-rs/crates/harness/src/lib.rs b/neurosploit-rs/crates/harness/src/lib.rs index 0faa555..aed84c6 100644 --- a/neurosploit-rs/crates/harness/src/lib.rs +++ b/neurosploit-rs/crates/harness/src/lib.rs @@ -43,6 +43,7 @@ pub mod replay; pub mod report; pub mod rl; pub mod sandbox; +pub mod sarif; pub mod scope; pub mod taint; pub mod transport; diff --git a/neurosploit-rs/crates/harness/src/pipeline.rs b/neurosploit-rs/crates/harness/src/pipeline.rs index 5563e68..1a073c2 100644 --- a/neurosploit-rs/crates/harness/src/pipeline.rs +++ b/neurosploit-rs/crates/harness/src/pipeline.rs @@ -607,6 +607,7 @@ const CHAIN_DOCTRINE: &str = "CHAIN THE FOOTHOLD (pivot to deeper, provable impa ยท A param that lands in a redirect/`Location` header โ†’ ALSO test CRLF/header injection on the SAME param (`%0d%0aX-Injected: pwned`, `%0d%0aSet-Cookie:`): an open redirect and response splitting share the sink, so never stop at the redirect.\n\ - Second-order & preconditions: a payload you STORE (profile/bio/name/review/filename) may only fire on a DIFFERENT page, often a privileged one (e.g. an admin search). Plant the payload, then TRIGGER it from every identity you hold; if the trigger page needs a role you lack, FIRST look for a privesc/IDOR/mass-assign to reach it, and if none exists, report the second-order as a CHAINED lead (payload stored + trigger located, blocked only by authorization) rather than dropping it.\n\ - Reuse loot relentlessly: every credential/JWT/cookie/API key/host you obtain is input to the next step โ€” carry it forward across modules and try it everywhere it might be accepted.\n\ +- Mine object references (this is the BOLA/IDOR engine): from EVERY response harvest each object identifier you see โ€” numeric ids, UUIDs, order/invoice/document/ticket numbers, account/customer ids, filenames, emails, and any signed/opaque token โ€” into a reference pool. Then systematically SUBSTITUTE another principal's identifier into every request that accepts one (path segment, query param, JSON field, header, cookie) and diff the response against your own: another user's data returned under your session IS the proof. Do this ACROSS identities โ€” register/hold โ‰ฅ2 accounts and cross them โ€” and across endpoints, because an id leaked on one route (a list/search/export) is often the key to an object on another (a detail/update/delete route). Enumerate sequential ids sparingly (a small benign sample, never a mass scrape) to show the pattern, then stop at proof.\n\ - Understand the BUSINESS & LOGIC: reason about what the app is FOR (payments, orders, tenancy, KYC, entitlements) and chain toward business impact โ€” payment/price/coupon abuse, cross-tenant data access, entitlement/limit bypass, workflow/state-machine skips (skip approval/verification steps), race conditions on balance/stock. These compound: each finding updates your model of the app for the next probe.\n\ - Stop at proof: demonstrate the impact with the SMALLEST safe step and report the CHAIN end-to-end; never destroy, overwrite, encrypt, mass-exfiltrate, or DoS to 'prove' it.\n\n"; @@ -2558,7 +2559,7 @@ async fn finish(cfg: RunConfig, _lib: &Library, pool: &ModelPool, recon: String, let _ = tx.send(n).await; } - // Coverage report (Strix-style): what was tested, how, and what was NOT โ€” + // Coverage report: what was tested, how, and what was NOT โ€” // so a reader can see the engagement's reach, not just its findings. write_coverage(&cfg, &selected, &findings); let artifacts = persist(&cfg, &recon, &transcript, &findings); diff --git a/neurosploit-rs/crates/harness/src/report.rs b/neurosploit-rs/crates/harness/src/report.rs index ee7af18..bc85abd 100644 --- a/neurosploit-rs/crates/harness/src/report.rs +++ b/neurosploit-rs/crates/harness/src/report.rs @@ -810,6 +810,7 @@ pub fn write_all(target: &str, findings: &[Finding], dir: &Path) -> std::io::Res md.push_str(&pocs_section(dir)); std::fs::write(dir.join("report.md"), md)?; std::fs::write(dir.join("report.json"), json_report(target, findings, &run_id, &meta))?; + std::fs::write(dir.join("report.sarif"), crate::sarif::to_string(target, findings))?; let pocs: Vec = std::fs::read_dir(dir.join("pocs")) .map(|rd| rd.filter_map(|e| e.ok()).map(|e| e.file_name().to_string_lossy().to_string()).collect()) .unwrap_or_default(); diff --git a/neurosploit-rs/crates/harness/src/sarif.rs b/neurosploit-rs/crates/harness/src/sarif.rs new file mode 100644 index 0000000..b64a9d4 --- /dev/null +++ b/neurosploit-rs/crates/harness/src/sarif.rs @@ -0,0 +1,248 @@ +//! SARIF 2.1.0 export for a finished run. +//! +//! SARIF (Static Analysis Results Interchange Format) is the format GitHub code +//! scanning, Azure DevOps, and most CI dashboards ingest. Emitting it next to +//! `report.md` lets a NeuroSploit run drop straight into a pipeline: upload +//! `report.sarif` and every finding shows up as an annotated alert, coloured by +//! severity, linked to its CWE, on the exact endpoint it was proven against. +//! +//! This is a pure projection of the findings already on disk โ€” no model calls, +//! no fabrication. It is emitted by [`crate::report::write_all`] and rebuilt by +//! [`crate::report::rebuild`], and can be regenerated on demand from the CLI. +//! +//! We follow the parts of the spec CI actually consumes: +//! - one `run` with a `tool.driver` carrying a de-duplicated `rules` array, +//! - `security-severity` (0.0..10.0) on each rule, which GitHub uses to bucket +//! the alert into critical/high/medium/low, +//! - one `result` per finding with `ruleId`, `level`, a message, and a +//! physical location pointing at the endpoint. + +use crate::types::Finding; +use serde_json::{json, Value}; +use std::collections::BTreeMap; + +const SCHEMA: &str = "https://raw.githubusercontent.com/oasis-tcs/sarif-spec/master/Schemata/sarif-schema-2.1.0.json"; + +/// SARIF result level. GitHub renders error/warning/note distinctly. +fn level_for(severity: &str) -> &'static str { + match severity.trim().to_ascii_lowercase().as_str() { + "critical" | "high" => "error", + "medium" => "warning", + "low" => "note", + _ => "none", + } +} + +/// Numeric CVSS base score parsed from the leading float of the `cvss` field +/// (e.g. "9.8 (AV:N/AC:L/...)" -> 9.8). Falls back to a severity-band midpoint +/// so GitHub still buckets the alert when no vector was graded. +fn security_severity(f: &Finding) -> String { + let parsed = f + .cvss + .trim() + .split(|c: char| !(c.is_ascii_digit() || c == '.')) + .find(|t| !t.is_empty()) + .and_then(|t| t.parse::().ok()) + .filter(|v| (0.0..=10.0).contains(v)); + let score = parsed.unwrap_or_else(|| match f.severity.trim().to_ascii_lowercase().as_str() { + "critical" => 9.5, + "high" => 7.5, + "medium" => 5.0, + "low" => 2.5, + _ => 0.0, + }); + format!("{score:.1}") +} + +/// A stable rule id for a finding: prefer the CWE, else the agent class. +fn rule_id(f: &Finding) -> String { + let cwe = f.cwe.trim(); + if cwe.is_empty() { + let agent = f.agent.trim(); + if agent.is_empty() { "NEUROSPLOIT.finding".to_string() } else { format!("NEUROSPLOIT.{agent}") } + } else if cwe.to_ascii_uppercase().starts_with("CWE-") { + cwe.to_ascii_uppercase() + } else { + format!("CWE-{cwe}") + } +} + +/// Help URI for a CWE-style rule id (deep link to the MITRE entry). +fn help_uri(rule_id: &str) -> Option { + let num: String = rule_id.chars().filter(|c| c.is_ascii_digit()).collect(); + if rule_id.starts_with("CWE-") && !num.is_empty() { + Some(format!("https://cwe.mitre.org/data/definitions/{num}.html")) + } else { + None + } +} + +/// Build the full SARIF 2.1.0 document for a run's findings. +pub fn to_sarif(target: &str, findings: &[Finding]) -> Value { + // De-duplicate rules by id; the first finding of each class defines the rule. + let mut rules: BTreeMap = BTreeMap::new(); + for f in findings { + let id = rule_id(f); + rules.entry(id.clone()).or_insert_with(|| { + let mut rule = json!({ + "id": id, + "name": if f.cwe.trim().is_empty() { f.agent.clone() } else { f.cwe.clone() }, + "shortDescription": { "text": rule_short_desc(f) }, + "defaultConfiguration": { "level": level_for(&f.severity) }, + "properties": { + "security-severity": security_severity(f), + "tags": rule_tags(f), + } + }); + if let Some(uri) = help_uri(&id) { + rule["helpUri"] = json!(uri); + } + rule + }); + } + + let results: Vec = findings.iter().map(|f| result_for(f)).collect(); + + json!({ + "$schema": SCHEMA, + "version": "2.1.0", + "runs": [{ + "tool": { + "driver": { + "name": "NeuroSploit", + "version": env!("CARGO_PKG_VERSION"), + "informationUri": "https://github.com/CyberSecurityUP/neurosploit-rs", + "rules": rules.into_values().collect::>(), + } + }, + "properties": { "target": target }, + "results": results, + }] + }) +} + +fn rule_short_desc(f: &Finding) -> String { + if !f.cwe.trim().is_empty() { + f.title.clone() + } else { + format!("{} finding", f.agent) + } +} + +/// Rule tags CI can facet on: OWASP + MITRE when known, plus "security". +fn rule_tags(f: &Finding) -> Vec { + let mut tags = vec!["security".to_string()]; + if !f.owasp.trim().is_empty() { tags.push(f.owasp.clone()); } + if !f.mitre.trim().is_empty() { tags.push(f.mitre.clone()); } + tags +} + +fn result_for(f: &Finding) -> Value { + // The endpoint is the closest thing to a physical location for a web + // finding; `location` (param/field/step) refines the message. + let uri = if f.endpoint.trim().is_empty() { "target".to_string() } else { f.endpoint.clone() }; + let mut text = f.title.clone(); + if !f.location.trim().is_empty() { + text.push_str(&format!(" โ€” {}", f.location)); + } + if !f.impact.trim().is_empty() { + text.push_str(&format!("\nImpact: {}", f.impact)); + } + + let mut props = json!({ + "confidence": f.confidence, + "validated": f.validated, + "severity": f.severity, + }); + if !f.cvss.trim().is_empty() { props["cvss"] = json!(f.cvss); } + if !f.owasp.trim().is_empty() { props["owasp"] = json!(f.owasp); } + if !f.mitre.trim().is_empty() { props["mitre"] = json!(f.mitre); } + if !f.review_status.trim().is_empty() { props["reviewStatus"] = json!(f.review_status); } + + json!({ + "ruleId": rule_id(f), + "level": level_for(&f.severity), + "message": { "text": text }, + "locations": [{ + "physicalLocation": { + "artifactLocation": { "uri": uri } + } + }], + "partialFingerprints": { "neurosploitFindingId": f.id }, + "properties": props, + }) +} + +/// Serialize the SARIF document to pretty JSON. +pub fn to_string(target: &str, findings: &[Finding]) -> String { + serde_json::to_string_pretty(&to_sarif(target, findings)).unwrap_or_else(|_| "{}".to_string()) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn finding(sev: &str, cwe: &str, cvss: &str) -> Finding { + Finding { + id: "f1".into(), + agent: "web".into(), + title: "SQL injection".into(), + severity: sev.into(), + cwe: cwe.into(), + cvss: cvss.into(), + endpoint: "https://app.example.com/api/login".into(), + ..Default::default() + } + } + + #[test] + fn levels_map_from_severity() { + assert_eq!(level_for("Critical"), "error"); + assert_eq!(level_for("high"), "error"); + assert_eq!(level_for("Medium"), "warning"); + assert_eq!(level_for("Low"), "note"); + assert_eq!(level_for("Info"), "none"); + } + + #[test] + fn cvss_score_parsed_from_vector_string() { + let f = finding("Critical", "CWE-89", "9.8 (AV:N/AC:L/PR:N/UI:N/S:U/C:H/I:H/A:H)"); + assert_eq!(security_severity(&f), "9.8"); + } + + #[test] + fn cvss_falls_back_to_severity_band() { + let f = finding("High", "CWE-89", ""); + assert_eq!(security_severity(&f), "7.5"); + } + + #[test] + fn rule_id_prefers_cwe() { + assert_eq!(rule_id(&finding("High", "CWE-89", "")), "CWE-89"); + assert_eq!(rule_id(&finding("High", "89", "")), "CWE-89"); + assert_eq!(rule_id(&finding("High", "", "")), "NEUROSPLOIT.web"); + } + + #[test] + fn help_uri_deep_links_cwe() { + assert_eq!(help_uri("CWE-89").as_deref(), Some("https://cwe.mitre.org/data/definitions/89.html")); + assert_eq!(help_uri("NEUROSPLOIT.web"), None); + } + + #[test] + fn document_is_wellformed_and_dedupes_rules() { + let fs = vec![ + finding("Critical", "CWE-89", "9.8"), + finding("High", "CWE-89", "8.1"), + finding("Medium", "CWE-79", "6.1"), + ]; + let doc = to_sarif("https://app.example.com", &fs); + assert_eq!(doc["version"], "2.1.0"); + let rules = doc["runs"][0]["tool"]["driver"]["rules"].as_array().unwrap(); + assert_eq!(rules.len(), 2, "two distinct CWEs -> two rules"); + let results = doc["runs"][0]["results"].as_array().unwrap(); + assert_eq!(results.len(), 3); + assert_eq!(results[0]["ruleId"], "CWE-89"); + assert_eq!(results[0]["level"], "error"); + } +}