{
 "license": "CC-BY-4.0",
 "attribution": "Fide AI, Agentic Cyber Explorer",
 "url": "https://agentic-cyber-explorer.pages.dev/events/gray-swan-caisi-aisi-indirect-injection-competition-2026/",
 "asOf": "2026-09-26",
 "id": "gray-swan-caisi-aisi-indirect-injection-competition-2026",
 "date": "2026-03-16",
 "datePrecision": "day",
 "title": "CAISI, UK AISI and Gray Swan competition finds concealed indirect injections succeed on all 13 frontier models",
 "lane": "defense",
 "kind": "paper",
 "summary": "A competition run by Gray Swan with NIST's CAISI, the UK AI Security Institute and frontier labs asked 464 participants to craft indirect prompt injections that make tool-use, coding and computer-use agents take harmful actions while hiding any sign of compromise from the user. Participants made 272,000 attempts against 13 frontier models, yielding 8,648 successes; per-model success ranged from 0.5% (Claude Opus 4.5) to 8.5% (Gemini 2.5 Pro), and at least one attack succeeded against every model.",
 "whyItMatters": "It adds concealment to the success criterion and finds that each of the 13 frontier models tested fell to at least one indirect injection.",
 "actors": [
  "gray-swan-ai",
  "us-caisi",
  "uk-aisi"
 ],
 "topics": [
  "prompt-injection",
  "capability-evaluation"
 ],
 "atlas": [
  "untrusted-content",
  "tools",
  "model"
 ],
 "artifacts": [
  "claude-opus-4",
  "gemini"
 ],
 "sources": [
  {
   "url": "https://arxiv.org/abs/2603.15714",
   "publisher": "arXiv",
   "title": "How Vulnerable Are AI Agents to Indirect Prompt Injections? Insights from a Large-Scale Public Competition",
   "date": "2026-03-16",
   "type": "primary",
   "accessed": "2026-09-25"
  },
  {
   "url": "https://www.nist.gov/blogs/caisi-research-blog/insights-ai-agent-security-large-scale-red-teaming-competition",
   "publisher": "NIST",
   "title": "Insights into AI Agent Security from a Large-Scale Red-Teaming Competition",
   "date": "2026-03-23",
   "type": "primary",
   "accessed": "2026-09-25"
  }
 ],
 "keyFacts": [
  {
   "fact": "464 participants, 272,000 attack attempts, 13 frontier models, 8,648 successful attacks.",
   "locator": "Abstract"
  },
  {
   "fact": "Per-model success ranged from 0.5% (Claude Opus 4.5) to 8.5% (Gemini 2.5 Pro); universal strategies transferred across 21 of 41 behaviors.",
   "locator": "Abstract"
  },
  {
   "fact": "CAISI reports at least one successful attack against every target model, and transfer tended to flow from more robust to less robust models.",
   "locator": "NIST blog, key findings"
  }
 ],
 "significance": 4,
 "fideQuestions": [],
 "methods": [
  "adaptive-red-teaming",
  "indirect-prompt-injection"
 ],
 "review": "assistant-drafted",
 "addedOn": "2026-09-25"
}