{
 "license": "CC-BY-4.0",
 "attribution": "Fide AI, Agentic Cyber Explorer",
 "url": "https://agentic-cyber-explorer.pages.dev/events/embrace-the-red-claude-code-auto-mode-bypass-2026/",
 "asOf": "2026-10-01",
 "id": "embrace-the-red-claude-code-auto-mode-bypass-2026",
 "date": "2026-08-26",
 "datePrecision": "day",
 "title": "Researcher reports a prompt-injection chain that gets code execution in Claude Code's Auto Mode; Anthropic closes it as working as designed",
 "lane": "attack",
 "kind": "vulnerability-disclosure",
 "summary": "Security researcher Johann Rehberger (Embrace The Red) reports that a request to summarize a web page led Claude Code with Opus 5 in Auto Mode to run attacker-controlled code in his lab setup, using a multi-step chain of individually benign-looking actions, with success in 3 of 5 to 4 of 5 trials per variant. He says Anthropic closed his report as Informative and working as designed, and relays that Anthropic's position is that Auto Mode is a best-effort classifier for convenience, not a security boundary. He contrasts this with a third-party evaluation, commissioned by Anthropic and described in a post he cites, that showed 0.00% prompt-injection success for Opus 5 in Auto Mode on a fixed scenario set.",
 "whyItMatters": "It is a researcher's small-sample targeted test of a shipped classifier-based approval mode, and the vendor response he reports places the security boundary at OS isolation and network egress control rather than at the classifier.",
 "actors": [
  "embrace-the-red",
  "anthropic",
  "trajectory-labs"
 ],
 "topics": [
  "prompt-injection",
  "jailbreaks-and-safeguards",
  "monitoring-and-control"
 ],
 "atlas": [
  "untrusted-content",
  "tools",
  "human-approver"
 ],
 "artifacts": [
  "claude-opus-5"
 ],
 "sources": [
  {
   "url": "https://embracethered.com/blog/posts/2026/breaking-claude-code-opus-5-and-automode/",
   "publisher": "Embrace The Red",
   "title": "Breaking Claude Code Opus 5 Auto Mode",
   "date": "2026-08-26",
   "type": "primary",
   "accessed": "2026-09-30"
  }
 ],
 "keyFacts": [
  {
   "fact": "The post says Auto Mode replaces human approval prompts with a safety classifier and has been the default starting mode in Claude Code since mid-August.",
   "locator": "Auto Mode Is Now the Default in Claude Code"
  },
  {
   "fact": "Bypass class: indirect prompt injection through attacker-controlled web content, which steers the agent, through actions that each look benign, into running code from an attacker-controlled directory; the agent's refusal to run an obviously malicious program did not prevent this. In the author's lab setup the code reached a command-and-control server he controlled.",
   "locator": "In A Nutshell; Walkthrough"
  },
  {
   "fact": "Reported success rates, small samples and not a universal measure: 3/5 (60%) for a variant with a remote stager and callback, 3/5 (60%) and 4/5 (80%) for two variants that launch a nested Claude Code instance. The author says rates improved as payloads were iterated with Codex and that results are representative of a motivated attacker but not comprehensive.",
   "locator": "Reliability and Attack Success Rate (ASR)"
  },
  {
   "fact": "In some runs the agent avoided compromise, for example by analyzing the archive statically or running its helper from a safe directory; in others it noticed the compromise afterward, and in a few runs Auto Mode denied the agent's command to stop the malicious process.",
   "locator": "Claude Realizes Attacks Too Late; Auto Mode Blocks Cleanup; Refusals"
  },
  {
   "fact": "Disclosure: the author says he emailed Anthropic's model bug bounty address without a response, then used its security reporting channel, and Anthropic closed the report as Informative and working as designed.",
   "locator": "Disclosure"
  },
  {
   "fact": "The author relays Anthropic's position as: Auto Mode is a convenience feature with a best-effort classifier, not a security guarantee, and the real boundary is OS isolation and network egress control. This is the researcher's account, not an Anthropic statement.",
   "locator": "Disclosure"
  },
  {
   "fact": "Per the post, a Trajectory Labs evaluation commissioned by Anthropic ran 72 indirect prompt-injection scenarios ten times each and showed 0.00% attack success for Opus 5 in Auto Mode; the author says his chain was not in that set. The evaluation itself was not reviewed for this record.",
   "locator": "Auto Mode Is Now the Default; The 0.00% Marketing Problem"
  },
  {
   "fact": "The author recommends running unattended coding agents in a container, VM or OS sandbox, restricting network egress, monitoring agents, and not exposing home directories or credentials.",
   "locator": "Mitigation: Sandboxing - Not Optional"
  }
 ],
 "significance": 3,
 "fideQuestions": [
  "FID-074",
  "FID-076"
 ],
 "methods": [
  "approval-bypass",
  "indirect-prompt-injection",
  "injection-classifiers"
 ],
 "review": "assistant-drafted",
 "addedOn": "2026-09-30"
}