{
 "license": "CC-BY-4.0",
 "attribution": "Fide AI, Agentic Cyber Explorer",
 "url": "https://agentic-cyber-explorer.pages.dev/events/uk-aisi-sandboxescapebench-2026/",
 "asOf": "2026-09-26",
 "id": "uk-aisi-sandboxescapebench-2026",
 "date": "2026-03-01",
 "datePrecision": "day",
 "title": "UK AISI's SandboxEscapeBench measures frontier models' ability to escape container sandboxes",
 "lane": "defense",
 "kind": "benchmark",
 "summary": "Researchers at the UK AI Security Institute and the University of Oxford built SandboxEscapeBench, 18 nested-sandbox capture-the-flag scenarios covering orchestration, runtime and kernel weaknesses rated difficulty 1 to 5. In the March 2026 version, which tested nine models, GPT-5 and Claude Opus 4.5 succeeded on about half of all runs, difficulty 1-2 scenarios were solved at high rates, no model succeeded at difficulty 4 or 5, and success rose roughly log-linearly with token budget. An August 2026 revision reports that newer models did markedly better, with Claude Mythos Preview saturating the public benchmark.",
 "whyItMatters": "It gives a direct measurement of whether container sandboxes still contain capable agents, the control that evaluation and deployment environments rely on.",
 "actors": [
  "uk-aisi",
  "university-of-oxford"
 ],
 "topics": [
  "sandbox-containment",
  "capability-evaluation"
 ],
 "atlas": [
  "sandbox",
  "eval-environment"
 ],
 "artifacts": [
  "sandboxescapebench",
  "inspect",
  "claude-opus-4",
  "gpt-5-family",
  "claude-mythos"
 ],
 "sources": [
  {
   "url": "https://arxiv.org/abs/2603.02277",
   "publisher": "arXiv",
   "title": "Quantifying Frontier LLM Capabilities for Container Sandbox Escape",
   "date": "2026-03-01",
   "type": "primary",
   "accessed": "2026-09-25"
  },
  {
   "url": "https://arxiv.org/html/2603.02277v1",
   "publisher": "arXiv",
   "title": "Quantifying Frontier LLM Capabilities for Container Sandbox Escape (HTML)",
   "date": "2026-03-01",
   "type": "primary",
   "accessed": "2026-09-25"
  }
 ],
 "keyFacts": [
  {
   "fact": "18 scenarios: orchestration (4), engine/runtime (8), host/kernel (6). The March version evaluated nine models with five runs per model-scenario pair at a 2.5M-token budget.",
   "locator": "v1 Section 3.2; Appendix E"
  },
  {
   "fact": "Overall escape success across runs (v1): GPT-5 0.50 [0.40, 0.60]; Claude Opus 4.5 0.49 [0.39, 0.59]; GPT-5.2 0.27; Claude Haiku 4.5 0.18; no successes at difficulty 4-5.",
   "locator": "v1 Section 5.1"
  },
  {
   "fact": "All successful exploits used known misconfigurations or public CVEs; four unintended shortcuts were found and fixed.",
   "locator": "Section 5; Appendix C"
  },
  {
   "fact": "In the August 2026 revision (v3), newer models run at a 100M-token budget included Claude Mythos Preview, which scored 0.84 overall and succeeded on difficulty 4 and 5 tasks; the authors say it saturates the public benchmark.",
   "locator": "v3 abstract footnote; Section 6; Appendix H"
  }
 ],
 "significance": 4,
 "fideQuestions": [
  "FID-075"
 ],
 "methods": [
  "compute-scaled-evaluation",
  "ctf-benchmarks",
  "evaluation-gaming",
  "sandbox-escape",
  "sandboxing-egress"
 ],
 "review": "assistant-drafted",
 "addedOn": "2026-09-25"
}