{
 "license": "CC-BY-4.0",
 "attribution": "Fide AI, Agentic Cyber Explorer",
 "url": "https://agentic-cyber-explorer.pages.dev/events/stanford-bountybench-2025/",
 "asOf": "2026-09-26",
 "id": "stanford-bountybench-2025",
 "date": "2025-05-21",
 "datePrecision": "day",
 "title": "BountyBench measures AI agents on detect, exploit and patch tasks from real bug bounties",
 "lane": "defense",
 "kind": "benchmark",
 "summary": "BountyBench, from Stanford-led researchers, builds 40 bug bounties across 25 real-world systems into 120 Detect, Exploit and Patch tasks with dollar values attached. In the first version the best Detect score was 5%, while OpenAI Codex CLI and Claude Code scored 90% and 87.5% on Patch, well above their Exploit scores. A July 2025 revision with more agents reported Codex CLI with o3-high at 12.5% on Detect and 90% on Patch.",
 "whyItMatters": "It puts offensive and defensive agent performance on the same real codebases and expresses results in bounty dollars.",
 "actors": [
  "stanford-university",
  "uc-berkeley"
 ],
 "topics": [
  "vulnerability-repair",
  "capability-evaluation",
  "vulnerability-discovery"
 ],
 "atlas": [
  "eval-environment"
 ],
 "artifacts": [
  "bountybench",
  "claude-sonnet",
  "openai-o-series"
 ],
 "sources": [
  {
   "url": "https://arxiv.org/abs/2505.15216",
   "publisher": "arXiv",
   "title": "BountyBench: Dollar Impact of AI Agent Attackers and Defenders on Real-World Cybersecurity Systems",
   "date": "2025-05-21",
   "type": "primary",
   "accessed": "2026-09-25"
  },
  {
   "url": "https://arxiv.org/abs/2505.15216v2",
   "publisher": "arXiv",
   "title": "BountyBench: Dollar Impact of AI Agent Attackers and Defenders on Real-World Cybersecurity Systems (v2)",
   "date": "2025-07-10",
   "type": "primary",
   "accessed": "2026-09-25"
  }
 ],
 "keyFacts": [
  {
   "fact": "25 systems, 40 bug bounties with awards from $10 to $30,485, covering 9 of the OWASP Top 10; 120 tasks.",
   "locator": "Abstract; Section 1"
  },
  {
   "fact": "v1 (21 May 2025, 5 agents): best Detect 5% (Codex CLI $2,400; Claude Code $1,350); Codex CLI 90% Patch ($14,422); Claude 3.7 Sonnet Thinking custom agent 67.5% Exploit; Codex CLI and Claude Code Patch 90% and 87.5% vs Exploit 32.5% and 57.5%.",
   "locator": "v1 abstract"
  },
  {
   "fact": "v2 (10 July 2025, 8 agents): Codex CLI o3-high 12.5% Detect ($3,720) and 90% Patch ($14,152); Codex CLI o4-mini 90% Patch ($14,422); Codex CLI o3-high, o4-mini and Claude Code scored 90%, 90% and 87.5% on Patch vs 47.5%, 32.5% and 57.5% on Exploit.",
   "locator": "v2 abstract"
  }
 ],
 "significance": 3,
 "fideQuestions": [
  "FID-075",
  "FID-088"
 ],
 "methods": [
  "automated-patching"
 ],
 "review": "assistant-drafted",
 "addedOn": "2026-09-25"
}