{
 "license": "CC-BY-4.0",
 "attribution": "Fide AI, Agentic Cyber Explorer",
 "url": "https://agentic-cyber-explorer.pages.dev/events/attacker-moves-second-adaptive-attacks-2025/",
 "asOf": "2026-09-26",
 "id": "attacker-moves-second-adaptive-attacks-2025",
 "date": "2025-10-10",
 "datePrecision": "day",
 "title": "'The Attacker Moves Second': adaptive attacks bypass 12 published jailbreak and injection defenses",
 "lane": "defense",
 "kind": "paper",
 "summary": "Nasr, Carlini, Tramèr and 11 co-authors apply gradient, reinforcement learning, search and human red-teaming attacks to 12 published defenses. Most defenses originally reported near-zero attack success, but the adaptive attacks exceed 90% success against most, and human red-teamers succeeded on every challenge in the subset of defenses they were given.",
 "whyItMatters": "It is the central evidence that static-benchmark robustness claims for prompt injection defenses do not hold against adaptive attackers.",
 "actors": [],
 "topics": [
  "prompt-injection",
  "jailbreaks-and-safeguards",
  "eval-validity"
 ],
 "atlas": [
  "untrusted-content",
  "model",
  "monitor"
 ],
 "artifacts": [
  "agentdojo",
  "spotlighting",
  "struq",
  "secalign"
 ],
 "sources": [
  {
   "url": "https://arxiv.org/abs/2510.09023",
   "publisher": "arXiv",
   "title": "The Attacker Moves Second: Stronger Adaptive Attacks Bypass Defenses Against Llm Jailbreaks and Prompt Injections",
   "date": "2025-10-10",
   "type": "primary",
   "accessed": "2026-09-25"
  },
  {
   "url": "https://arxiv.org/html/2510.09023",
   "publisher": "arXiv",
   "title": "The Attacker Moves Second (HTML)",
   "date": "2025-10-10",
   "type": "primary",
   "accessed": "2026-09-25"
  },
  {
   "url": "https://simonwillison.net/2025/Nov/2/new-prompt-injection-papers/",
   "publisher": "Simon Willison's Weblog",
   "title": "New prompt injection papers: Agents Rule of Two and The Attacker Moves Second",
   "date": "2025-11-02",
   "type": "secondary",
   "accessed": "2026-09-25"
  }
 ],
 "keyFacts": [
  {
   "fact": "12 defenses bypassed with attack success above 90% for most; most had originally reported near-zero ASR.",
   "locator": "Abstract"
  },
  {
   "fact": "Spotlighting and Prompt Sandwiching on AgentDojo: as low as 1% attack success under the benchmark's static attacks (authors' re-implementation) vs over 95% with the adaptive search attack.",
   "locator": "Section 5.1"
  },
  {
   "fact": "Meta SecAlign (AgentDojo): 2% originally vs 96% adaptive; PromptGuard and Protect AI detector: over 90%; PIGuard: 71%; MELON: 76%, rising to 95% with full defense knowledge.",
   "locator": "Sections 5.2-5.4"
  },
  {
   "fact": "A human red-teaming competition with 500+ participants and a $20,000 prize pool succeeded in every challenge on the subset of defenses included in the human study.",
   "locator": "Section 6; Appendix E"
  }
 ],
 "significance": 5,
 "fideQuestions": [
  "FID-075"
 ],
 "methods": [
  "adaptive-red-teaming",
  "injection-classifiers",
  "input-delimiting",
  "instruction-priority-training",
  "jailbreaking"
 ],
 "review": "assistant-drafted",
 "addedOn": "2026-09-25"
}