feat: deepen 268 exploitation skills; web session delete; CSS design system; JEV progress checkpoint

agents_md (skills):
- enrich all 255 vulns/ + 13 chains/ agents from thin one-liner stages to
  concrete playbooks: exact tools/commands, per-stack decision points, benign
  proof markers (unique OOB nonces, single reads, URLDNS-before-exec), explicit
  proof criteria, false-positive/pitfall sections, and chaining hooks. Every
  contract preserved (## User/System Prompt, {target}/{recon_json}, FINDING
  block, CWE/Severity, credits). avg 37->53 lines; loader parses all 449.

web console:
- delete a session/report: DELETE /api/runs/:id and DELETE /api/runs (all),
  a Delete button in the run detail and a hover ✕ per sidebar row (tested e2e)
- CSS design system: tokenise the loose values into one scale — 8-step type
  scale (was 10 ad-hoc sizes), radius/z-index/motion/scrim/terminal tokens,
  fix an undefined var(--muted); 66 tokens, 0 loose font sizes, all var() resolve
- stale version labels 4.0.0/4.2.0 -> 4.2.1

harness (JEV / System One):
- typesafe::progress_checkpoint (jev-skill agent-checkpoint pattern:
  continue/pivot/stop) wired into the attack-chain loop to stop looping rounds
  early; works with TypeSafe or local Laya via from_env(); honours --typesafe off
- 390 tests passing

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
CyberSecurityUPandClaude Opus 4.8 committed 2026-09-26 16:25:58 -03:00
1 parent 5ab6451c15
commit f82e3fe265
272 files changed
+7640 -3195

No files matched your search

@@ -1432,6 +1432,7 @@ async fn attack_chain(pool: &ModelPool, cfg: &RunConfig, recon: &str,
let mut all_new: Vec<Finding> = Vec::new();
let mut loot: Vec<String> = Vec::new();
let mut round_summaries: Vec<String> = Vec::new();
let mut seen: std::collections::HashSet<String> = confirmed.iter().map(finding_key).collect();
// Frontier = footholds to expand this round; start with confirmed, best-first.
@@ -1479,6 +1480,25 @@ async fn attack_chain(pool: &ModelPool, cfg: &RunConfig, recon: &str,
break;
}
all_new.extend(validated.clone());
round_summaries.push(format!("round {round}: +{} validated finding(s), {} loot item(s) total", validated.len(), loot.len()));
// JEV / System One progress checkpoint (the jev-skill agent-checkpoint
// pattern): from the 2nd round on, ask a calibrated backend — TypeSafe
// or the local Laya shim — whether recent rounds are still productive,
// and stop early on a clear loop instead of burning the remaining depth.
// Optional: only when a decision backend is active and --typesafe ≠ off.
if round >= 2 && round < max_rounds
&& !std::env::var("NEUROSPLOIT_TYPESAFE").unwrap_or_default().trim().eq_ignore_ascii_case("off")
{
if let Some(ts) = crate::typesafe::TypeSafe::from_env() {
let objective = cfg.objective.clone().unwrap_or_else(|| "maximise proven, chained impact".into());
if let Ok((label, p_continue)) = ts.progress_checkpoint(&objective, &round_summaries, round, max_rounds).await {
if label == "stop" && p_continue < 0.5 {
let _ = tx.send(format!("⛓ System One checkpoint ({}): rounds are looping (p_continue {:.2}) — stopping chain early", ts.backend_label(), p_continue)).await;
break;
}
}
}
}
// Next round expands the freshly-validated footholds, best-first.
frontier = validated;
frontier.sort_by_key(|f| std::cmp::Reverse(sev_rank(&f.severity)));
@@ -252,6 +252,33 @@ impl TypeSafe {
Ok(ans.get("inject").and_then(|x| x.noul).unwrap_or(0.0))
}
/// Agent progress checkpoint (the jev-skill "goal-drift / stuck-loop"
/// pattern): given the objective and the last few round summaries, is the
/// engagement still making progress on THIS foothold, or is it looping /
/// drifting and better spent elsewhere? A calibrated Choice the chain loop
/// can branch on instead of always burning every remaining round.
/// Returns (label ∈ {continue, pivot, stop}, p_continue).
pub async fn progress_checkpoint(&self, objective: &str, recent: &[String], round: usize, max: usize) -> Result<(String, f64), String> {
let mut qs = BTreeMap::new();
qs.insert("progress".to_string(), Question::choice(
"Given the objective and the recent round summaries, what is the best next move for the autonomous agent on THIS foothold?",
&[
("continue", "recent rounds produced new footholds/loot/impact — pressing here is paying off"),
("pivot", "progress stalled here, but the loot/knowledge gathered opens a clearly better direction"),
("stop", "the last rounds repeat the same actions/observations with no new impact — a loop; stop spending rounds here"),
],
));
let state = serde_json::json!({
"objective": objective,
"round": round,
"rounds_max": max,
"recent_rounds": recent.iter().rev().take(4).rev().map(|s| s.chars().take(400).collect::<String>()).collect::<Vec<_>>(),
});
let ans = self.evaluate(state, qs).await?;
let a = ans.get("progress").cloned().unwrap_or_default();
Ok((a.choice.clone().unwrap_or_else(|| "continue".into()), a.p("continue")))
}
pub async fn adjudicate(&self, state: serde_json::Value) -> Result<Adjudication, String> {
let mut qs = BTreeMap::new();
qs.insert(
@@ -396,6 +423,22 @@ mod tests {
assert!(!clear.wants_review());
}
#[test]
fn progress_checkpoint_answer_parses_to_a_decision() {
// The chain loop reads (label, p_continue) from a Choice over
// continue/pivot/stop — the jev-skill agent-checkpoint shape.
let raw = r#"{
"answers": {
"progress": {"choice":"stop","probabilities":{"continue":0.12,"pivot":0.2,"stop":0.68},"confidence":0.7}
}
}"#;
let parsed: ApiResponse = serde_json::from_str(raw).unwrap();
let a = &parsed.answers["progress"];
assert_eq!(a.choice.as_deref(), Some("stop"));
assert!(a.p("continue") < 0.5, "a stalled loop should not read as continue");
assert!(a.p("stop") > a.p("continue"));
}
#[test]
fn no_key_means_no_client() {
// Deterministic only when the var is actually unset in the test env.