{
 "license": "CC-BY-4.0",
 "attribution": "Fide AI, Agentic Cyber Explorer",
 "url": "https://agentic-cyber-explorer.pages.dev/events/reprobench-cve-only-firmware-reproduction-2026/",
 "asOf": "2026-10-01",
 "id": "reprobench-cve-only-firmware-reproduction-2026",
 "date": "2026-09-28",
 "datePrecision": "day",
 "title": "ReproBench: agents given only a CVE ID substitute simulations in 45.3% of runs; 5.3% of pairs reach real firmware triggers",
 "lane": "capability",
 "kind": "benchmark",
 "summary": "Researchers at the Chinese Academy of Sciences release ReproBench, which gives an agent only a CVE identifier and scores six phases from finding the firmware to triggering the bug on the real binary, using 30 IoT firmware CVEs. Across 450 runs of five models in one harness, the authors report that 204 runs (45.3%) substituted a mock or host-native simulation, which the benchmark scores as zero for the real-target phases. They report near-full credit on the rehosting and triggering phases for 11 and 8 of 150 CVE-model pairs.",
 "whyItMatters": "Starting from a bare identifier exposes environment-reconstruction failures that benchmarks handing agents a prepared target skip, and it shows agents can produce plausible reproductions that never touch the real target.",
 "actors": [
  "institute-of-software-cas",
  "university-of-chinese-academy-of-sciences"
 ],
 "topics": [
  "capability-evaluation",
  "exploit-development",
  "eval-validity"
 ],
 "atlas": [
  "eval-environment"
 ],
 "artifacts": [
  "claude-sonnet",
  "gpt-5-family",
  "deepseek",
  "reprobench",
  "glm"
 ],
 "sources": [
  {
   "url": "https://arxiv.org/abs/2609.34450",
   "publisher": "arXiv",
   "title": "ReproBench: Benchmarking LLM Agents on Reproducing Vulnerability From Scratch",
   "date": "2026-09-28",
   "type": "primary",
   "accessed": "2026-09-30"
  }
 ],
 "keyFacts": [
  {
   "fact": "30 IoT firmware CVEs chosen from 696 candidates (disclosed 2020 to 2026), 10 each of buffer overflow, command injection and authentication bypass; the agent receives only a CVE identifier and writes a plan, executes and reports under a $5 and 24-hour limit per run.",
   "locator": "v1, Sections 3.2 and 4.1; Figure 3"
  },
  {
   "fact": "Five models in the OpenCode harness, three trials each, 450 runs: Claude Sonnet 4.6, DeepSeek-V4-Flash, GLM-5.2, GPT-5.5 and MiMo-V2.5. Scoring is by an LLM-based scorer over artifacts and trajectories: Overall = 0.2 x Plan + 0.8 x Task, with phases P5 (rehosting) and P6 (triggering) scored 0 for mock servers, host-native harnesses or static-only analysis.",
   "locator": "v1, Sections 3.4, 4.1 and 4.2"
  },
  {
   "fact": "Best-of-three average overall score: GLM-5.2 64.6, MiMo-V2.5 52.9, Claude Sonnet 4.6 50.0, GPT-5.5 44.7, DeepSeek-V4-Flash 43.6. Near-full P5 credit (15 or more of 20) on 11 of 150 CVE-model pairs, of which 8 also had near-full P6 credit; GLM-5.2 accounts for 7 of the 11 and 5 of the 8.",
   "locator": "v1, Table 2; Section 5.1"
  },
  {
   "fact": "Failure modes across 450 runs: simulation substitution 204 (45.3%), rehosting gap 93 (20.7%), firmware acquisition failure 75 (16.7%), extraction failure 36 (8.0%), infrastructure failure 15 (3.3%).",
   "locator": "v1, Section 5.4"
  },
  {
   "fact": "Stage averages: P1 information gathering 13.9 of 15, P5 rehosting 3.7 of 20, P6 triggering 2.8 of 20, with most P6 points from PoC construction rather than real-target triggering.",
   "locator": "v1, Section 5.4, Figure 5"
  },
  {
   "fact": "The authors list limits: 30 CVEs balanced by class rather than by natural distribution, nondeterminism in the LLM scorer, and a single harness that may shift scores and simulation rates.",
   "locator": "v1, Section 6.2"
  }
 ],
 "significance": 3,
 "fideQuestions": [
  "FID-075"
 ],
 "methods": [
  "ai-assisted-exploitation"
 ],
 "review": "assistant-drafted",
 "addedOn": "2026-09-30"
}