{
 "license": "CC-BY-4.0",
 "attribution": "Fide AI, Agentic Cyber Explorer",
 "url": "https://agentic-cyber-explorer.pages.dev/findings/frontier-models-produce-working-exploits/",
 "asOf": "2026-09-26",
 "id": "frontier-models-produce-working-exploits",
 "claim": "On ExploitGym (May 2026), the strongest agents produced working exploits for 157 and 120 of 898 instances with mitigations off; with standard mitigations on, 45 and 21 survived.",
 "evidenceKind": "measured",
 "scope": "One benchmark with an LLM judge, run with lab safeguards disabled; the top model (Claude Mythos Preview) was unreleased, and most other models fell to zero with mitigations on.",
 "topics": [
  "exploit-development",
  "capability-evaluation"
 ],
 "atlas": [],
 "evidence": [
  {
   "event": "exploitgym-benchmark-2026",
   "note": "Table 3: Mythos Preview 157 and GPT-5.5 120 of 898. Table 5: with mitigations re-enabled, 45 and 21 remained."
  }
 ],
 "relations": [],
 "statusHistory": [
  {
   "status": "reported",
   "on": "2026-05-11",
   "why": "ExploitGym results.",
   "event": "exploitgym-benchmark-2026",
   "kind": "evidence"
  },
  {
   "status": "qualified",
   "on": "2026-09-08",
   "why": "Later work shows cyber benchmark scores depend heavily on pipeline choices.",
   "event": "benchmark-scores-pipeline-dependent-cyber-2026",
   "kind": "evidence"
  },
  {
   "status": "qualified",
   "on": "2026-09-25",
   "why": "Correction: the pipeline audit covered eight knowledge and multiple-choice benchmarks, not ExploitGym. The qualification rests on ExploitBench, where no publicly deployed model reached code execution on V8.",
   "event": "exploitbench-benchmark-2026",
   "kind": "correction"
  }
 ],
 "halfLifeDays": 240,
 "wouldChange": "Replication under a standardized pipeline.",
 "fideQuestions": [],
 "methods": [
  "ai-assisted-exploitation",
  "ctf-benchmarks"
 ],
 "review": "assistant-drafted",
 "addedOn": "2026-09-25"
}