{
 "license": "CC-BY-4.0",
 "attribution": "Fide AI, Agentic Cyber Explorer",
 "url": "https://agentic-cyber-explorer.pages.dev/events/cyberpersistbench-installation-persistence-2026/",
 "asOf": "2026-10-01",
 "id": "cyberpersistbench-installation-persistence-2026",
 "date": "2026-09-29",
 "datePrecision": "day",
 "title": "CyberPersistBench scores agent post-compromise persistence: 27.6% to 42.4% on 203 tasks, 5.5% to 13.3% under native defenses",
 "lane": "capability",
 "kind": "benchmark",
 "summary": "Researchers at Shanghai AI Laboratory release CyberPersistBench, which starts agents from a restricted foothold and scores whether they keep durable access after credentials are revoked and services or hosts are disrupted. It has 203 single-host tasks in seven mechanism categories, a 65-task multi-host extension and a 128-task subset with native security controls, scored on six levels. The authors report pass@3 success of 27.6% to 42.4% for five models on the common scaffold and 5.5% to 13.3% with defenses enabled.",
 "whyItMatters": "It measures a stage that exploit-focused cyber benchmarks stop before, and its stage-wise scores locate where agents fail rather than only whether they succeed.",
 "actors": [
  "shanghai-ai-laboratory"
 ],
 "topics": [
  "capability-evaluation",
  "ai-enabled-intrusion"
 ],
 "atlas": [
  "eval-environment"
 ],
 "artifacts": [
  "inspect",
  "gpt-5-family",
  "claude-sonnet",
  "deepseek",
  "kimi",
  "cyberpersistbench",
  "glm"
 ],
 "sources": [
  {
   "url": "https://arxiv.org/abs/2609.36573",
   "publisher": "arXiv",
   "title": "CyberPersistBench: Evaluating LLM-Based Cyber Attackers on Installation and Persistence",
   "date": "2026-09-29",
   "type": "primary",
   "accessed": "2026-09-30"
  }
 ],
 "keyFacts": [
  {
   "fact": "203 core tasks in 7 categories and 46 mechanism families, with a 65-task multi-host extension and a 128-task active-defense subset; tasks run in Inspect AI containers or VMs with hidden verification, fresh challenge nonces and host-integrity checks. Reference solutions were written with AI coding assistance and manually validated.",
   "locator": "v1, Sections 3.2 to 3.4"
  },
  {
   "fact": "Six-level prefix scoring: construction, native establishment, baseline activation, access independence, operational recovery and reinitialization persistence; final success also requires satisfying the host-integrity contract.",
   "locator": "v1, Section 3.3; Table 1"
  },
  {
   "fact": "Core suite, Inspect ReAct scaffold, pass@3: GPT-5.6-sol 86/203 (42.4%), Claude Sonnet 5 82/203 (40.4%), DeepSeek-V4-pro 77/203 (37.9%), GLM-5.2 64/203 (31.5%), Kimi K2.6 56/203 (27.6%).",
   "locator": "v1, Table 1"
  },
  {
   "fact": "The abstract's 44.8% upper bound is GPT-5.6-sol in Codex CLI at 81/181; Claude Sonnet 5 in Claude Code scores 86/197 (43.7%). Both denominators exclude tasks blocked in all three attempts by the scaffold's safety policy, so they are not comparable with the 203-task rows.",
   "locator": "v1, Table 1 caption; Section 4.2"
  },
  {
   "fact": "The largest drop is from constructing an artifact to registering it natively (Level 1 to Level 2), 11.8 to 22.7 percentage points across configurations; integrity satisfaction ranges from 64.0% to 80.7%, and final success is below the Level 6 rate for every configuration.",
   "locator": "v1, Section 4.2, Figure 5"
  },
  {
   "fact": "On the 128 defense-enabled tasks (Inspect ReAct, pass@3), success with defenses off versus on: GPT-5.6-sol 40.6% to 13.3%, DeepSeek-V4-pro 40.6% to 10.9%, GLM-5.2 34.4% to 10.2%, Claude Sonnet 5 33.6% to 8.6%, Kimi K2.6 18.8% to 5.5%.",
   "locator": "v1, Table 2"
  },
  {
   "fact": "Multi-host extension (65 tasks), success: GPT-5.6-sol 29.2%, GLM-5.2 27.7%, Claude Sonnet 5 18.5%, DeepSeek-V4-pro 16.9%, Kimi K2.6 10.8%.",
   "locator": "v1, Table 3"
  },
  {
   "fact": "Each attempt was capped at 40 agent turns and 160 messages under a shared system prompt describing an authorized cyber range; the authors say the benchmark does not measure end-to-end attacks and models no adaptive defenders.",
   "locator": "v1, Appendix D; Appendix A (Limitations)"
  }
 ],
 "significance": 3,
 "fideQuestions": [
  "FID-075"
 ],
 "methods": [
  "cyber-ranges"
 ],
 "review": "assistant-drafted",
 "addedOn": "2026-09-30"
}