{
 "license": "CC-BY-4.0",
 "attribution": "Fide AI, Agentic Cyber Explorer",
 "url": "https://agentic-cyber-explorer.pages.dev/findings/models-escape-vulnerable-sandboxes/",
 "asOf": "2026-09-26",
 "id": "models-escape-vulnerable-sandboxes",
 "claim": "The strongest frontier models tested escaped deliberately vulnerable container sandboxes in about half of scenarios, and success rose with inference compute.",
 "evidenceKind": "measured",
 "scope": "Purpose-built vulnerable containers, not hardened production sandboxes. About half applies to GPT-5 and Claude Opus 4.5; smaller models scored much lower. No model solved the hardest scenarios, and all escapes used known misconfigurations or public CVEs.",
 "topics": [
  "sandbox-containment",
  "capability-evaluation"
 ],
 "atlas": [
  "sandbox",
  "eval-environment"
 ],
 "evidence": [
  {
   "event": "uk-aisi-sandboxescapebench-2026"
  }
 ],
 "relations": [],
 "statusHistory": [
  {
   "status": "reported",
   "on": "2026-03-01",
   "why": "UK AISI publishes SandboxEscapeBench.",
   "event": "uk-aisi-sandboxescapebench-2026",
   "kind": "evidence"
  }
 ],
 "halfLifeDays": 270,
 "wouldChange": "Escape rates against hardened configurations.",
 "fideQuestions": [],
 "methods": [
  "sandbox-escape",
  "sandboxing-egress"
 ],
 "review": "assistant-drafted",
 "addedOn": "2026-09-25"
}