{
 "license": "CC-BY-4.0",
 "attribution": "Fide AI, Agentic Cyber Explorer",
 "url": "https://agentic-cyber-explorer.pages.dev/events/anthropic-opus-4-6-system-card-prompt-injection-2026/",
 "asOf": "2026-09-26",
 "id": "anthropic-opus-4-6-system-card-prompt-injection-2026",
 "date": "2026-02-05",
 "datePrecision": "day",
 "title": "Claude Opus 4.6 system card reports prompt injection rates by surface, attempts and safeguards",
 "lane": "defense",
 "kind": "system-card",
 "summary": "Anthropic's Claude Opus 4.6 system card reports prompt injection attack success separately for tool use (Gray Swan's ART benchmark), coding and computer use (Gray Swan's Shade adaptive attacker), and browser use (an internal Best-of-N attacker), with and without extra safeguards and across different attempt budgets. For Opus 4.6, results range from 0% in coding to 85.7% in computer use with 200 attempts and no safeguards (78.6% with extended thinking). Anthropic notes that, unlike earlier Claude models, extended thinking increased ART attack success for this model.",
 "whyItMatters": "It is an unusually detailed lab disclosure of agent prompt injection rates, and it shows that robustness depends strongly on the surface, the attacker's budget and the safeguards.",
 "actors": [
  "anthropic",
  "gray-swan-ai"
 ],
 "topics": [
  "prompt-injection",
  "eval-validity"
 ],
 "atlas": [
  "untrusted-content",
  "tools",
  "model"
 ],
 "artifacts": [
  "agent-red-teaming-benchmark",
  "shade-arena",
  "claude-opus-4",
  "claude-sonnet"
 ],
 "sources": [
  {
   "url": "https://www-cdn.anthropic.com/6a5fa276ac68b9aeb0c8b6af5fa36326e0e166dd/Claude%20Opus%204.6%20System%20Card.pdf",
   "publisher": "Anthropic",
   "title": "System Card: Claude Opus 4.6",
   "date": "2026-02",
   "type": "primary",
   "accessed": "2026-09-25"
  },
  {
   "url": "https://venturebeat.com/security/prompt-injection-measurable-security-metric-one-ai-developer-publishes-numbers",
   "publisher": "VentureBeat",
   "title": "Anthropic published the prompt injection failure rates that enterprise security teams have been asking every vendor for",
   "date": "2026-02-10",
   "type": "secondary",
   "accessed": "2026-09-25"
  }
 ],
 "keyFacts": [
  {
   "fact": "ART benchmark (Gray Swan, 19 scenarios) at k=100: Opus 4.6 14.8% without extended thinking, 21.7% with it; Opus 4.5 16.5%.",
   "locator": "Section 5.2.1, Figure 5.2.1.A"
  },
  {
   "fact": "Coding (Shade adaptive attacker): Opus 4.6 0.0% ASR at 1 and 200 attempts, with or without safeguards; Sonnet 4.5 standard thinking 87.5% at 200 attempts without safeguards.",
   "locator": "Section 5.2.2.1, Table 5.2.2.1.A"
  },
  {
   "fact": "Computer use (stronger Shade attacker), Opus 4.6 without safeguards: 17.8% (1 attempt) and 78.6% (200) with extended thinking, 20.0% and 85.7% with standard thinking; with safeguards and extended thinking: 9.7% and 57.1%.",
   "locator": "Section 5.2.2.2, Table 5.2.2.2.A"
  },
  {
   "fact": "Browser use (389 scenarios, 10 attack strings each), Opus 4.6 extended thinking without safeguards: 2.06% of scenarios, 0.29% of attempts. Standard thinking with safeguards: 0.26% and 0.03% with the previous safeguards, 0.77% and 0.08% with the updated safeguards.",
   "locator": "Section 5.2.2.3, Tables 5.2.2.3.A-B"
  }
 ],
 "significance": 4,
 "fideQuestions": [
  "FID-075"
 ],
 "methods": [
  "adaptive-red-teaming",
  "indirect-prompt-injection"
 ],
 "review": "assistant-drafted",
 "addedOn": "2026-09-25"
}