diff --git a/.env.example b/.env.example index 14688f2..edbf72a 100755 --- a/.env.example +++ b/.env.example @@ -68,3 +68,10 @@ NOUS_API_KEY= # model through it as `--model litellm:`. #LITELLM_BASE_URL=http://localhost:4000/v1 LITELLM_API_KEY= + +# TypeSafe System One (RLCD — Reinforcement Learning for Calibrated Decisions). +# When set, NeuroSploit adjudicates each finding with a calibrated verdict +# (confirmed/needs-review/rejected) + an "impact demonstrated" judgment over the +# EVIDENCE, refining confidence and the needs-review boundary. Additive — never +# overrules a deterministic validator. Disable with NEUROSPLOIT_TYPESAFE=off. +TYPESAFE_API_KEY= diff --git a/README.md b/README.md index ea7de1b..3b28917 100755 --- a/README.md +++ b/README.md @@ -525,6 +525,21 @@ neurosploit provenance scan report.pdf.txt # is this ours? which build? neurosploit provenance verify runs/ns-… # manifest vs findings ``` +### TypeSafe System One — calibrated adjudication (RLCD) + +When `TYPESAFE_API_KEY` is set, each finding is adjudicated by TypeSafe's +System One model (Jev) — a **calibrated decision** over its *evidence*, not its +prose: a `Choice` of `{confirmed, needs-review, rejected}` with a probability +distribution, plus a `Noul` on whether real impact was demonstrated. The result +refines the finding's confidence and moves borderline cases to needs-review. + +It is **additive**: a deterministic validator still rules (a rejected finding +stays rejected), and TypeSafe can only lower confidence or flag for review, +never resurrect a claim. Every adjudication is written to the audit trail. +Disable with `NEUROSPLOIT_TYPESAFE=off`. This is the RLCD (Reinforcement +Learning for Calibrated Decisions) tier of the model stack — typed judgments +where the harness needs a number, not a paragraph. + ### Scope-evasion resistance, evidence integrity, untrusted output Three hardening passes, all enforced in code: diff --git a/neurosploit-rs/crates/harness/src/lib.rs b/neurosploit-rs/crates/harness/src/lib.rs index adf6b7d..9293c52 100644 --- a/neurosploit-rs/crates/harness/src/lib.rs +++ b/neurosploit-rs/crates/harness/src/lib.rs @@ -47,6 +47,7 @@ pub mod scope; pub mod taint; pub mod transport; pub mod types; +pub mod typesafe; pub mod uncertainty; pub mod validation; pub mod waf; diff --git a/neurosploit-rs/crates/harness/src/pipeline.rs b/neurosploit-rs/crates/harness/src/pipeline.rs index d9a1b3d..79a68a9 100644 --- a/neurosploit-rs/crates/harness/src/pipeline.rs +++ b/neurosploit-rs/crates/harness/src/pipeline.rs @@ -27,6 +27,41 @@ pub struct RunOutput { pub denied: Option, } +/// Build the TypeSafe "state" for a finding — its EVIDENCE, not its prose. +/// +/// System One judges what it is given; giving it the model's own narrative +/// would just launder the narrative back as a probability. So the state is the +/// recorded request/response facts and the structured fields, and nothing the +/// agent wrote about them. +fn typesafe_state(f: &Finding) -> serde_json::Value { + let ex = |x: &Option| -> serde_json::Value { + match x { + Some(e) => serde_json::json!({ + "method": e.method, "url": e.url, "status": e.status, + "content_type": e.content_type, + "body_snippet": e.body.chars().take(1200).collect::(), + }), + None => serde_json::Value::Null, + } + }; + let evidence = f.evidence_data.as_ref().map(|ev| serde_json::json!({ + "baseline": ex(&ev.baseline), + "attack": ex(&ev.attack), + "marker": ev.marker, + "marker_observed": ev.marker_observed, + "browser_executed": ev.browser_executed, + "callback_received": ev.callback_received, + })).unwrap_or(serde_json::Value::Null); + serde_json::json!({ + "class": f.cwe, + "title": f.title, + "endpoint": f.endpoint, + "payload": f.payload, + "auth_context": f.auth_context, + "evidence": evidence, + }) +} + /// A run that stopped before it started. /// /// Returned when the egress or the authorization boundary refuses the target. @@ -2157,6 +2192,57 @@ async fn finish(cfg: RunConfig, _lib: &Library, pool: &ModelPool, recon: String, let _ = tx.send(format!("notify: 🧾 {}", crate::integrity::summary(&audits))).await; } + // TypeSafe System One adjudication (optional). When TYPESAFE_API_KEY is set, + // a calibrated Choice over {confirmed, needs-review, rejected} plus an + // "impact demonstrated" Noul is asked over each finding's EVIDENCE (never + // its prose). Additive: it never overrules a deterministic validator — + // evidence still rules — it only refines confidence and moves a borderline + // finding to needs-review. Off with NEUROSPLOIT_TYPESAFE=off. + if std::env::var("NEUROSPLOIT_TYPESAFE").unwrap_or_default().trim().to_lowercase() != "off" { + if let Some(ts) = crate::typesafe::TypeSafe::from_env() { + let _ = tx.send("notify: 🧮 TypeSafe System One adjudicating findings…".to_string()).await; + let mut refined = 0usize; + for f in findings.iter_mut() { + let state = typesafe_state(f); + match ts.adjudicate(state).await { + Ok(adj) => { + // A deterministic rejection is final; TypeSafe can only + // lower confidence or flag for review, never resurrect. + if f.review_status != "rejected" { + let cal = adj.calibrated_confidence(); + // Take the more conservative of the two confidences. + f.confidence = f.confidence.min(cal.max(0.05)); + if adj.wants_review() && f.review_status.is_empty() { + f.review_status = "needs-review".into(); + f.validated = false; + f.review_reason = format!( + "TypeSafe: p(confirmed)={:.2}, impact={:.2} — below the bar for auto-confirm", + adj.p_confirmed, adj.impact_demonstrated + ); + } + refined += 1; + } + audit.append( + crate::audit::AuditRecord::new("typesafe", "adjudicate-finding", &f.endpoint) + .hypothesis(&f.id) + .decision(&format!("{}: p_confirmed={:.2} impact={:.2} conf={:.2}", adj.verdict, adj.p_confirmed, adj.impact_demonstrated, adj.confidence)) + .tool("typesafe:jev-latest") + .capability(&cap_id) + .result(&f.title), + ); + } + Err(e) => { + let _ = tx.send(format!("notify: ⚠ TypeSafe unavailable ({e}) — keeping deterministic verdicts")).await; + break; // don't hammer a failing service + } + } + } + if refined > 0 { + let _ = tx.send(format!("notify: 🧮 TypeSafe refined {refined} finding(s) with calibrated confidence")).await; + } + } + } + // PoC re-validation: re-run each finding's recorded proof and demote any // that no longer reproduces. This is the harness checking its own work — a // bug that was hotfixed between discovery and reporting, or a "proof" that diff --git a/neurosploit-rs/crates/harness/src/typesafe.rs b/neurosploit-rs/crates/harness/src/typesafe.rs new file mode 100644 index 0000000..fdcb078 --- /dev/null +++ b/neurosploit-rs/crates/harness/src/typesafe.rs @@ -0,0 +1,306 @@ +//! TypeSafe — System One judgments as programming primitives. +//! +//! Most of the harness's expensive calls ask a text model a question that is +//! really a *decision*: is this finding confirmed, needs-review or rejected? +//! how severe is it? did the evidence actually demonstrate impact? A text model +//! answers in prose the harness then has to parse, and the answer is not +//! calibrated — "high confidence" is a word, not a number. +//! +//! TypeSafe's System One model ([Jev](https://docs.typesafe.ai)) answers those +//! as typed judgments with calibrated probabilities instead of text: +//! +//! ```text +//! state (the finding's evidence) + a typed question +//! │ +//! ▼ +//! Choice → one option + a probability distribution + confidence +//! Score → a position on an ordered scale + probabilities +//! Noul → probability a condition holds (0.0–1.0) +//! ``` +//! +//! This is the right shape for adjudication and gating: a `Choice` over +//! `{confirmed, needs-review, rejected}` gives the pipeline a calibrated number +//! to gate on rather than a parsed adjective. It is **additive and optional** — +//! it never overrides a deterministic validator (evidence still rules), only +//! sharpens the confidence and the needs-review boundary. Enabled only when +//! `TYPESAFE_API_KEY` is set; absent, everything behaves exactly as before. +//! +//! Contract per the live docs: `POST https://api.typesafe.ai/v1/systemone`, +//! `Authorization: Bearer `, `model: "jev-latest"`, `state` + +//! `questions` map; each answer carries the primitive's typed result. + +use serde::{Deserialize, Serialize}; +use std::collections::BTreeMap; +use std::time::Duration; + +const ENDPOINT: &str = "https://api.typesafe.ai/v1/systemone"; +const MODEL: &str = "jev-latest"; + +/// A question to evaluate against the state. +#[derive(Debug, Clone, Serialize)] +#[serde(tag = "type", rename_all = "lowercase")] +pub enum Question { + /// Pick one of a defined set. Criteria: option -> description (or null). + Choice { + instructions: String, + criteria: BTreeMap>, + }, + /// Whether a condition holds. Returns the probability of "true". + Noul { + instructions: String, + criteria: NoulCriteria, + }, + /// A position on an ordered scale of described levels. + Score { + instructions: String, + criteria: Vec, + }, +} + +#[derive(Debug, Clone, Serialize)] +pub struct NoulCriteria { + #[serde(rename = "true")] + pub yes: String, + #[serde(rename = "false")] + pub no: String, +} + +impl Question { + pub fn choice(instructions: &str, options: &[(&str, &str)]) -> Question { + let criteria = options.iter().map(|(k, v)| ((*k).to_string(), if v.is_empty() { None } else { Some((*v).to_string()) })).collect(); + Question::Choice { instructions: instructions.into(), criteria } + } + pub fn noul(instructions: &str, yes: &str, no: &str) -> Question { + Question::Noul { instructions: instructions.into(), criteria: NoulCriteria { yes: yes.into(), no: no.into() } } + } + pub fn score(instructions: &str, levels: &[&str]) -> Question { + Question::Score { instructions: instructions.into(), criteria: levels.iter().map(|s| s.to_string()).collect() } + } +} + +#[derive(Debug, Clone, Serialize)] +struct Request { + model: String, + state: serde_json::Value, + questions: BTreeMap, +} + +/// One typed answer. Only the fields relevant to the primitive are populated. +#[derive(Debug, Clone, Deserialize, Default)] +pub struct Answer { + /// Choice: the selected option. + #[serde(default)] + pub choice: Option, + /// Score: the probability-weighted value. + #[serde(default)] + pub score: Option, + /// Noul: the probability the condition holds (0..1). + #[serde(default)] + pub noul: Option, + /// Choice/Score: per-option probability distribution. + #[serde(default)] + pub probabilities: BTreeMap, + /// Choice/Score: how concentrated the distribution is (not correctness). + #[serde(default)] + pub confidence: Option, +} + +impl Answer { + /// Probability mass on a named option (0.0 if absent). + pub fn p(&self, option: &str) -> f64 { + self.probabilities.get(option).copied().unwrap_or(0.0) + } +} + +#[derive(Debug, Clone, Deserialize)] +struct ApiResponse { + #[serde(default)] + answers: BTreeMap, +} + +/// A TypeSafe client. Cheap to construct; holds the key and an HTTP client. +#[derive(Clone)] +pub struct TypeSafe { + key: String, + client: reqwest::Client, +} + +impl TypeSafe { + /// Build from `TYPESAFE_API_KEY`. None when unset — the caller then skips + /// System One entirely rather than failing. + pub fn from_env() -> Option { + let key = std::env::var("TYPESAFE_API_KEY").ok().filter(|k| !k.trim().is_empty())?; + Some(TypeSafe { + key, + client: reqwest::Client::builder().timeout(Duration::from_secs(30)).build().unwrap_or_default(), + }) + } + + pub fn new(key: &str) -> TypeSafe { + TypeSafe { key: key.to_string(), client: reqwest::Client::new() } + } + + /// Evaluate a set of independent questions over one state, in parallel (the + /// API runs them together — they cannot see one another's answers). + pub async fn evaluate(&self, state: serde_json::Value, questions: BTreeMap) -> Result, String> { + let req = Request { model: MODEL.into(), state, questions }; + // A short retry on the documented transient codes (429/529). + let mut attempt = 0; + loop { + attempt += 1; + let resp = self.client.post(ENDPOINT).bearer_auth(&self.key).json(&req).send().await; + match resp { + Ok(r) => { + let status = r.status().as_u16(); + if matches!(status, 429 | 529) && attempt < 3 { + tokio::time::sleep(Duration::from_millis(400 * attempt as u64)).await; + continue; + } + if !r.status().is_success() { + let body = r.text().await.unwrap_or_default(); + return Err(format!("typesafe HTTP {status}: {}", body.chars().take(200).collect::())); + } + let parsed: ApiResponse = r.json().await.map_err(|e| format!("typesafe response parse: {e}"))?; + return Ok(parsed.answers); + } + Err(e) if attempt < 3 => { + tokio::time::sleep(Duration::from_millis(400 * attempt as u64)).await; + let _ = e; + } + Err(e) => return Err(format!("typesafe request failed: {e}")), + } + } + } + + /// Adjudicate one finding: a calibrated `{confirmed, needs-review, rejected}` + /// judgment over its evidence, plus a "was impact demonstrated" Noul. The + /// state is the finding's own recorded facts — never the model's prose about + /// it — so the judgment is over evidence, not narrative. + pub async fn adjudicate(&self, state: serde_json::Value) -> Result { + let mut qs = BTreeMap::new(); + qs.insert( + "verdict".to_string(), + Question::choice( + "Given ONLY the recorded request/response evidence in the state, does it deterministically demonstrate the claimed vulnerability class?", + &[ + ("confirmed", "the evidence demonstrates the class beyond reasonable doubt"), + ("needs-review", "plausible but the evidence is incomplete — a human should decide"), + ("rejected", "the evidence does not support the claim, or contradicts it"), + ], + ), + ); + qs.insert( + "impact_demonstrated".to_string(), + Question::noul( + "Does the evidence show REAL impact (data read/written, code executed, a boundary crossed), as opposed to only that a payload was reflected or an error appeared?", + "concrete impact is shown in the evidence", + "no impact is shown — only a mechanic or a reflection", + ), + ); + let answers = self.evaluate(state, qs).await?; + let verdict = answers.get("verdict").cloned().unwrap_or_default(); + let impact = answers.get("impact_demonstrated").and_then(|a| a.noul).unwrap_or(0.0); + Ok(Adjudication { + verdict: verdict.choice.clone().unwrap_or_else(|| "needs-review".into()), + p_confirmed: verdict.p("confirmed"), + p_needs_review: verdict.p("needs-review"), + p_rejected: verdict.p("rejected"), + confidence: verdict.confidence.unwrap_or(0.0), + impact_demonstrated: impact, + }) + } +} + +/// The calibrated result of adjudicating a finding. +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct Adjudication { + pub verdict: String, + pub p_confirmed: f64, + pub p_needs_review: f64, + pub p_rejected: f64, + /// Distribution concentration — NOT correctness (per TypeSafe's docs). + pub confidence: f64, + /// Probability real impact was shown (0..1). + pub impact_demonstrated: f64, +} + +impl Adjudication { + /// A calibrated confidence for the finding: the probability it is confirmed, + /// tempered by whether impact was actually demonstrated. Bounded 0..1. + pub fn calibrated_confidence(&self) -> f64 { + (self.p_confirmed * (0.5 + 0.5 * self.impact_demonstrated)).clamp(0.0, 1.0) + } + /// Should this go to human review? Low separation between confirmed and the + /// alternatives, or a rejected-leaning verdict on a claimed-confirmed one. + pub fn wants_review(&self) -> bool { + self.verdict == "needs-review" || (self.p_confirmed < 0.6 && self.p_rejected < 0.6) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn questions_serialize_to_the_documented_shape() { + let q = Question::choice("Which?", &[("a", "first"), ("b", "")]); + let v = serde_json::to_value(&q).unwrap(); + assert_eq!(v["type"], "choice"); + assert_eq!(v["criteria"]["a"], "first"); + assert!(v["criteria"]["b"].is_null(), "an empty description serializes as null per the API"); + + let n = Question::noul("Holds?", "yes it does", "no it doesn't"); + let nv = serde_json::to_value(&n).unwrap(); + assert_eq!(nv["type"], "noul"); + assert_eq!(nv["criteria"]["true"], "yes it does"); + assert_eq!(nv["criteria"]["false"], "no it doesn't"); + + let s = Question::score("Rate", &["low", "mid", "high"]); + let sv = serde_json::to_value(&s).unwrap(); + assert_eq!(sv["type"], "score"); + assert_eq!(sv["criteria"][2], "high"); + } + + #[test] + fn answers_parse_and_expose_probabilities() { + let raw = r#"{ + "answers": { + "verdict": {"choice":"confirmed","probabilities":{"confirmed":0.82,"needs-review":0.13,"rejected":0.05},"confidence":0.77}, + "impact_demonstrated": {"noul":0.9} + } + }"#; + let parsed: ApiResponse = serde_json::from_str(raw).unwrap(); + let v = &parsed.answers["verdict"]; + assert_eq!(v.choice.as_deref(), Some("confirmed")); + assert!((v.p("confirmed") - 0.82).abs() < 1e-9); + assert_eq!(v.p("absent-option"), 0.0); + assert_eq!(parsed.answers["impact_demonstrated"].noul, Some(0.9)); + } + + #[test] + fn calibrated_confidence_folds_in_demonstrated_impact() { + // High p_confirmed but NO demonstrated impact → confidence is held back. + let a = Adjudication { verdict: "confirmed".into(), p_confirmed: 0.9, p_needs_review: 0.05, p_rejected: 0.05, confidence: 0.8, impact_demonstrated: 0.0 }; + assert!((a.calibrated_confidence() - 0.45).abs() < 1e-9, "no impact halves the weight"); + + // Same, with full impact → near p_confirmed. + let b = Adjudication { impact_demonstrated: 1.0, ..a.clone() }; + assert!((b.calibrated_confidence() - 0.9).abs() < 1e-9); + } + + #[test] + fn review_is_wanted_on_a_split_distribution() { + let split = Adjudication { verdict: "confirmed".into(), p_confirmed: 0.45, p_needs_review: 0.3, p_rejected: 0.25, confidence: 0.4, impact_demonstrated: 0.5 }; + assert!(split.wants_review(), "no option clears 0.6 — a human should look"); + let clear = Adjudication { verdict: "confirmed".into(), p_confirmed: 0.88, p_needs_review: 0.08, p_rejected: 0.04, confidence: 0.8, impact_demonstrated: 0.9 }; + assert!(!clear.wants_review()); + } + + #[test] + fn no_key_means_no_client() { + // Deterministic only when the var is actually unset in the test env. + if std::env::var("TYPESAFE_API_KEY").is_err() { + assert!(TypeSafe::from_env().is_none()); + } + } +}