mirror of
https://github.com/CyberSecurityUP/NeuroSploit.git
synced 2026-10-07 00:17:53 +02:00
v4.2.2: validation keeps reproducible findings, /pocs command, version bump
Field feedback: real, reproducible findings (missing HSTS, insecure cookie flags, internal IP leaked in a header) were being down-rated/dropped by the adversarial opinion-vote. The user's rule: never discard something real; another person must be able to reproduce the same finding. Validation: - has_http_receipt(): a finding whose proof is a captured HTTP response (status line / security headers / Set-Cookie / a header the class is proven by) or a file:line citation is REPRODUCIBLE by definition. - validate(): a finding unanimously rejected by the vote but carrying such a receipt is NO LONGER dropped — it is kept as needs-review (capped to what the receipt alone proves), because "the response lacks HSTS" is a fact, not a story. grounded_receipt() now also recognizes an in-prose HTTP receipt. - reproducibility: a kept finding at a URL with no explicit repro steps gets a minimal pasteable `curl -i` so anyone can reproduce the exact finding. Also: - /pocs (aliases /poc /evidence /artifacts): list a run's synthesized/written PoC + evidence files with their paths (was: "unknown command /pocs"). - version bumped to 4.2.2 across CLI/clap/web; stale v4.2.1/v4.1.0 labels fixed. 423 tests passing. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
1 parent
45ad7bf359
commit
b55778d6f6
9 files changed
+141
-34
No files matched your search
Generated
+2
-2
@@ -940,7 +940,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "neurosploit"
|
||||
version = "4.2.1"
|
||||
version = "4.2.2"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"clap",
|
||||
@@ -957,7 +957,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "neurosploit-harness"
|
||||
version = "4.2.1"
|
||||
version = "4.2.2"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"base64",
|
||||
|
||||
@@ -3,7 +3,7 @@ members = ["crates/harness", "app"]
|
||||
resolver = "2"
|
||||
|
||||
[workspace.package]
|
||||
version = "4.2.1"
|
||||
version = "4.2.2"
|
||||
edition = "2021"
|
||||
license = "MIT"
|
||||
repository = "https://github.com/JoasASantos/NeuroSploit"
|
||||
|
||||
@@ -13,8 +13,8 @@ use std::path::{Path, PathBuf};
|
||||
#[command(
|
||||
name = "neurosploit",
|
||||
version,
|
||||
about = "NeuroSploit v4.2.1 — multi-model autonomous pentest harness",
|
||||
long_about = "NeuroSploit v4.2.1 — a Rust multi-model harness that drives a pool of LLMs \
|
||||
about = "NeuroSploit v4.2.2 — multi-model autonomous pentest harness",
|
||||
long_about = "NeuroSploit v4.2.2 — a Rust multi-model harness that drives a pool of LLMs \
|
||||
(API key or local subscription: Claude/Codex/Gemini/Grok/OpenCode/Hermes) to autonomously test a target. \
|
||||
After recon it INTELLIGENTLY selects only the agents matching the discovered surface, runs \
|
||||
them in parallel, then validates every finding by cross-model voting before reporting.\n\n\
|
||||
|
||||
@@ -148,7 +148,7 @@ struct LiveCheckpoint {
|
||||
pub(crate) const ACCEPTED: &[&str] = &[
|
||||
"/?", "/agents", "/attach", "/audit", "/auth", "/burp", "/cap", "/capability", "/chain", "/changed", "/clear", "/config",
|
||||
"/context", "/continue", "/creds", "/diff", "/exclude", "/exit", "/expand", "/feed",
|
||||
"/finding", "/findings", "/focus", "/forget", "/full", "/go", "/goal", "/graph", "/guardrail", "/guardrails", "/help",
|
||||
"/pocs", "/poc", "/evidence", "/artifacts", "/finding", "/findings", "/focus", "/forget", "/full", "/go", "/goal", "/graph", "/guardrail", "/guardrails", "/help",
|
||||
"/history", "/idle", "/inscope", "/instructions", "/integration", "/integrations", "/key", "/log",
|
||||
"/authorize", "/grant", "/inscope-set", "/scope-file", "/scopefile", "/import-scope", "/authorization", "/authz", "/program", "/logs", "/mcp", "/memory", "/model", "/models", "/objective", "/objectives", "/observe",
|
||||
"/observe-only", "/offline",
|
||||
@@ -164,7 +164,7 @@ const COMMANDS: &[&str] = &[
|
||||
"/help", "/onboard", "/show", "/config", "/providers", "/model", "/key", "/sub", "/target",
|
||||
"/scope-file", "/authorization", "/class", "/repo", "/auth", "/creds", "/focus", "/objective", "/scope-out", "/attach", "/context", "/mcp", "/offline",
|
||||
"/class", "/research", "/quick", "/economy", "/eco", "/votes", "/chain", "/recon", "/tempmail", "/timeout", "/proxy", "/burp", "/ua", "/agents", "/only", "/theme", "/clear", "/run", "/stop", "/pause", "/continue", "/runs", "/results", "/report",
|
||||
"/status", "/logs", "/diff", "/retest", "/validate", "/finding", "/expand", "/integrations",
|
||||
"/status", "/logs", "/diff", "/retest", "/validate", "/finding", "/pocs", "/expand", "/integrations",
|
||||
"/memory", "/forget", "/graph", "/inscope", "/observe", "/guardrail", "/policy",
|
||||
"/capability", "/audit", "/quit",
|
||||
];
|
||||
@@ -1277,6 +1277,31 @@ pub async fn repl(base: &Path, auth: SessionAuth) -> anyhow::Result<()> {
|
||||
if live_now { println!(" \x1b[2m(run still streaming in background — /logs for what happened while browsing)\x1b[0m"); }
|
||||
}
|
||||
}
|
||||
"/pocs" | "/poc" | "/evidence" | "/artifacts" => {
|
||||
// List the PoC and evidence artifacts a run produced (synthesized
|
||||
// or agent-written), with their paths so they can be retrieved.
|
||||
let h = history.lock().unwrap();
|
||||
let rec = if arg.trim().is_empty() { h.last() } else { pick(&h, arg) };
|
||||
match rec {
|
||||
None => println!(" no run yet — /runs to list, or /pocs <id> after a run"),
|
||||
Some(r) if r.workdir.is_empty() => println!(" run #{} has no workdir on record", r.id),
|
||||
Some(r) => {
|
||||
let dir = std::path::Path::new(&r.workdir);
|
||||
let mut any = false;
|
||||
for (sub, label) in [("pocs", "PoC scripts"), ("evidence", "evidence")] {
|
||||
let p = dir.join(sub);
|
||||
let mut files: Vec<String> = std::fs::read_dir(&p).map(|rd| rd.filter_map(|e| e.ok()).map(|e| e.file_name().to_string_lossy().to_string()).collect()).unwrap_or_default();
|
||||
files.sort();
|
||||
if !files.is_empty() {
|
||||
any = true;
|
||||
println!(" \x1b[1m{label}\x1b[0m ({}):", p.display());
|
||||
for f in files { println!(" {}", p.join(&f).display()); }
|
||||
}
|
||||
}
|
||||
if !any { println!(" no pocs/ or evidence/ files for run #{} ({})", r.id, dir.display()); }
|
||||
}
|
||||
}
|
||||
}
|
||||
"/finding" | "/findings" => {
|
||||
// Build the finding pool: live run if active, else a past run.
|
||||
let pool: Vec<Finding> = match &active {
|
||||
|
||||
@@ -1935,26 +1935,35 @@ async fn validate(candidates: Vec<Finding>, pool: &ModelPool, sys: &str, vote_n:
|
||||
f.review_status = "needs-review".into();
|
||||
f.review_reason = format!("voter rejected the narrative, mechanic retained: {}", f.review_reason.trim());
|
||||
f.confidence = f.confidence.min(0.5);
|
||||
} else if grounded_receipt(&f) {
|
||||
// Unanimously rejected, but the MECHANISM was demonstrated —
|
||||
// a real engagement rejected "no rate limiting on the reset
|
||||
// flow" because the agent claimed email flooding and only
|
||||
// proved that 25 requests went through unthrottled. The
|
||||
// claim was inflated; the measurement was real, and
|
||||
// discarding it hid a genuine gap from the report.
|
||||
//
|
||||
// So an over-claimed finding is capped and flagged rather
|
||||
// than deleted: the reader gets the fact, not the story
|
||||
// that was built on it.
|
||||
} else if grounded_receipt(&f) || has_http_receipt(&f) {
|
||||
// Unanimously rejected by the opinion-vote, but the finding
|
||||
// carries a concrete, REPRODUCIBLE receipt (a captured HTTP
|
||||
// response, a header/cookie the class is proven by, or a
|
||||
// file:line citation). Another person can reproduce this
|
||||
// exact finding with one request — so it is NOT dropped.
|
||||
// The adversarial validator is tuned to reject low-impact and
|
||||
// "theoretical" issues, but "the response lacks HSTS" or "the
|
||||
// cookie has no Secure flag" is a FACT, not a story. We cap
|
||||
// the severity to what the receipt alone proves and flag it
|
||||
// for human review, never delete it.
|
||||
let cap = "Low";
|
||||
let was = f.severity.clone();
|
||||
f.severity = cap.to_string();
|
||||
if sev_rank(&f.severity) < sev_rank(cap) { f.severity = cap.to_string(); }
|
||||
f.review_status = "needs-review".into();
|
||||
f.review_reason = format!(
|
||||
"impact not demonstrated — capped from {was} to {cap}. Validator: {}",
|
||||
f.review_reason.trim()
|
||||
);
|
||||
f.confidence = f.confidence.min(0.5);
|
||||
f.review_reason = if was == f.severity {
|
||||
format!("kept — reproducible receipt present; impact not independently demonstrated. Validator: {}", f.review_reason.trim())
|
||||
} else {
|
||||
format!("kept & capped from {was} to {} — reproducible receipt present, impact not demonstrated. Validator: {}", f.severity, f.review_reason.trim())
|
||||
};
|
||||
f.confidence = f.confidence.max(0.3).min(0.6);
|
||||
}
|
||||
// Reproducibility: a kept finding that reaches a URL endpoint but
|
||||
// carries no explicit repro steps gets a minimal, pasteable one,
|
||||
// so another person can reproduce the exact finding.
|
||||
if (f.validated || f.review_status == "needs-review")
|
||||
&& f.repro_steps.is_empty()
|
||||
&& (f.endpoint.starts_with("http://") || f.endpoint.starts_with("https://")) {
|
||||
f.repro_steps = vec![format!("curl -i -s '{}' # inspect the response (status + headers + body) that proves this finding", f.endpoint)];
|
||||
}
|
||||
let label = if f.validated { "CONFIRMED" } else if f.review_status == "needs-review" { "needs-review" } else { "rejected" };
|
||||
let _ = txc.send(format!("vote {} → {} ({})", f.title, label, f.votes)).await;
|
||||
@@ -2154,9 +2163,39 @@ fn grounded_receipt(f: &Finding) -> bool {
|
||||
if f.review_reason.contains("receipt_missing") || f.votes.contains("receipt_missing") {
|
||||
return false;
|
||||
}
|
||||
if has_http_receipt(f) {
|
||||
return true;
|
||||
}
|
||||
crate::grounding::ground(f, "", crate::grounding::GroundMode::Either).ok
|
||||
}
|
||||
|
||||
/// True when the finding's evidence carries a concrete, reproducible HTTP
|
||||
/// receipt — a captured response (status line / headers / Set-Cookie) or a
|
||||
/// `file:line` code citation. This is the test for "another person can
|
||||
/// reproduce this exact finding": if the proof is the response itself (a missing
|
||||
/// security header, an insecure cookie flag, an internal IP leaked in a header,
|
||||
/// a status code), it is reproducible with a single request and must never be
|
||||
/// discarded by an opinion-based vote — demoted to needs-review at worst.
|
||||
fn has_http_receipt(f: &Finding) -> bool {
|
||||
if f.evidence_data.is_some() { return true; }
|
||||
let hay = format!("{}\n{}", f.evidence, f.payload).to_lowercase();
|
||||
// A captured HTTP response or the headers/fields these deterministic classes
|
||||
// are proven by.
|
||||
const SIGNS: &[&str] = &[
|
||||
"http/1.1", "http/2", "http/1.0", "status: ", "status code", "status=",
|
||||
"set-cookie", "strict-transport-security", "x-frame-options",
|
||||
"content-security-policy", "access-control-allow-origin", "location:",
|
||||
"server:", "www-authenticate", "< http", "=> http", "response:", "200 ok",
|
||||
"301 ", "302 ", "401 ", "403 ", "404 ", "500 ", "curl ",
|
||||
];
|
||||
if SIGNS.iter().any(|s| hay.contains(s)) { return true; }
|
||||
// A white-box file:line citation is also a reproducible receipt.
|
||||
f.endpoint.contains(':') && f.endpoint.chars().any(|c| c.is_ascii_digit())
|
||||
&& (f.endpoint.contains(".rs") || f.endpoint.contains(".py") || f.endpoint.contains(".js")
|
||||
|| f.endpoint.contains(".php") || f.endpoint.contains(".java") || f.endpoint.contains(".go")
|
||||
|| f.endpoint.contains(".ts") || f.endpoint.contains(".rb"))
|
||||
}
|
||||
|
||||
/// Adversarial refutation pass: every confirmed **High/Critical** finding is
|
||||
/// re-examined by a skeptical panel that tries to prove it's a false positive.
|
||||
/// A finding that fails to withstand a majority of skeptics is dropped. Lower
|
||||
@@ -4440,6 +4479,34 @@ mod extraction_tests {
|
||||
assert!(!looks_like_refusal("No vulnerabilities were found in the tested endpoints."));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_response_backed_finding_is_a_reproducible_receipt_and_never_dropped() {
|
||||
// Missing HSTS: the proof is the response headers — reproducible with one
|
||||
// request, must never be discarded by an opinion-vote.
|
||||
let hsts = Finding {
|
||||
title: "No Strict-Transport-Security header".into(),
|
||||
severity: "Low".into(), cwe: "CWE-319".into(),
|
||||
endpoint: "https://scapi.rockstargames.com/".into(),
|
||||
evidence: "HTTP/2 200\nserver: cloudflare\n(no strict-transport-security header present)".into(),
|
||||
..Default::default()
|
||||
};
|
||||
assert!(has_http_receipt(&hsts), "a captured response IS a receipt");
|
||||
// Insecure cookie flags — Set-Cookie in evidence.
|
||||
let cookie = Finding {
|
||||
title: "Cookie without Secure/HttpOnly".into(), severity: "Low".into(), cwe: "CWE-614".into(),
|
||||
endpoint: "https://scapi.rockstargames.com/".into(),
|
||||
evidence: "Set-Cookie: bal=1; path=/ (no Secure, no HttpOnly, no SameSite)".into(),
|
||||
..Default::default()
|
||||
};
|
||||
assert!(has_http_receipt(&cookie));
|
||||
// Pure prose with no receipt is NOT a reproducible receipt.
|
||||
let vague = Finding { title: "Maybe vulnerable".into(), evidence: "the app seems insecure".into(), ..Default::default() };
|
||||
assert!(!has_http_receipt(&vague));
|
||||
// A white-box file:line citation IS a receipt.
|
||||
let wb = Finding { title: "SQLi".into(), endpoint: "src/db.py:42".into(), evidence: "query = f\"...{id}\"".into(), ..Default::default() };
|
||||
assert!(has_http_receipt(&wb));
|
||||
}
|
||||
|
||||
/// The bare-array happy path for agent selection.
|
||||
#[test]
|
||||
fn a_string_array_parses_plain() {
|
||||
|
||||
Reference in new issue
Block a user