{
 "license": "CC-BY-4.0",
 "attribution": "Fide AI, Agentic Cyber Explorer",
 "url": "https://agentic-cyber-explorer.pages.dev/findings/crash-is-not-exploitation/",
 "asOf": "2026-09-26",
 "id": "crash-is-not-exploitation",
 "claim": "Counting a crash as exploitation success overstates capability; most public models in May 2026 stalled before code execution on a browser engine.",
 "evidenceKind": "measured",
 "scope": "One benchmark on one target class.",
 "topics": [
  "eval-validity",
  "exploit-development"
 ],
 "atlas": [],
 "evidence": [
  {
   "event": "exploitbench-benchmark-2026",
   "note": "Of 41 V8 bugs, crashes were common but no publicly deployed model reached arbitrary code execution in the primary arm; the unreleased Mythos Preview did on 18."
  }
 ],
 "relations": [
  {
   "type": "qualifies",
   "target": "frontier-models-produce-working-exploits",
   "note": "On hardened V8 targets, publicly deployed models rarely got past primitives to code execution; code execution at scale came only from an unreleased model."
  }
 ],
 "statusHistory": [
  {
   "status": "reported",
   "on": "2026-05-13",
   "why": "ExploitBench separates crashes from exploitation progress.",
   "event": "exploitbench-benchmark-2026",
   "kind": "evidence"
  }
 ],
 "halfLifeDays": 240,
 "wouldChange": "A later measurement showing models routinely reach code execution.",
 "fideQuestions": [],
 "methods": [
  "ai-assisted-exploitation",
  "ctf-benchmarks"
 ],
 "review": "assistant-drafted",
 "addedOn": "2026-09-25"
}