{
 "license": "CC-BY-4.0",
 "attribution": "Fide AI, Agentic Cyber Explorer",
 "url": "https://agentic-cyber-explorer.pages.dev/events/exploitbench-benchmark-2026/",
 "asOf": "2026-09-26",
 "id": "exploitbench-benchmark-2026",
 "date": "2026-05-13",
 "datePrecision": "day",
 "title": "ExploitBench grades AI exploit development as a 16-step capability ladder on V8 bugs",
 "lane": "capability",
 "kind": "benchmark",
 "summary": "Carnegie Mellon researchers released ExploitBench, which scores exploitation progress on 41 V8 JavaScript-engine vulnerabilities across 16 flags from reaching the bug through arbitrary read/write, control-flow hijack and code execution. The paper reports that public models routinely reach and crash vulnerable code but rarely achieve arbitrary code execution, while one private frontier model succeeded on roughly half of cases.",
 "whyItMatters": "Graded scoring separates reaching or crashing a bug from building a working exploit, which crash-as-success benchmarks conflate.",
 "actors": [
  "carnegie-mellon-university",
  "bugcrowd"
 ],
 "topics": [
  "exploit-development",
  "capability-evaluation",
  "eval-validity"
 ],
 "atlas": [
  "eval-environment"
 ],
 "artifacts": [
  "exploitbench"
 ],
 "sources": [
  {
   "url": "https://arxiv.org/abs/2605.14153",
   "publisher": "arXiv",
   "title": "ExploitBench: A Capability Ladder Benchmark for LLM Cybersecurity Agents",
   "date": "2026-05-13",
   "type": "primary",
   "accessed": "2026-09-25"
  }
 ],
 "keyFacts": [
  {
   "fact": "41 V8 vulnerabilities; 16 measurable capability flags; 8 public frontier models plus 1 private model evaluated.",
   "locator": "Abstract"
  },
  {
   "fact": "The private frontier model reached arbitrary code execution on approximately half of cases.",
   "locator": "Abstract"
  }
 ],
 "significance": 3,
 "fideQuestions": [
  "FID-075"
 ],
 "methods": [
  "ai-assisted-exploitation"
 ],
 "review": "assistant-drafted",
 "addedOn": "2026-09-25"
}