{
 "license": "CC-BY-4.0",
 "attribution": "Fide AI, Agentic Cyber Explorer",
 "url": "https://agentic-cyber-explorer.pages.dev/findings/lab-defenses-reduce-not-eliminate/",
 "asOf": "2026-09-26",
 "id": "lab-defenses-reduce-not-eliminate",
 "claim": "Frontier labs' layered defenses reduce prompt injection in tool-use, browser and computer-use agents but do not eliminate it.",
 "evidenceKind": "reported",
 "scope": "Rates are self-reported by the labs that built the systems; no independent adaptive test of 2026 production defenses is recorded.",
 "topics": [
  "prompt-injection",
  "jailbreaks-and-safeguards"
 ],
 "atlas": [
  "untrusted-content",
  "model"
 ],
 "evidence": [
  {
   "event": "deepmind-gemini-ipi-lessons-2025",
   "note": "Gemini 2.5 tool-use scenarios (email, calendar); adversarial training cut TAP success from 99.8% to 53.6% in the email scenario."
  },
  {
   "event": "anthropic-claude-in-chrome-pilot-pi-2025",
   "note": "11.2% of internal red-team cases after mitigations."
  },
  {
   "event": "anthropic-browser-use-pi-mitigations-2025",
   "note": "About 1% adaptive-attacker success reported."
  },
  {
   "event": "openai-atlas-rl-automated-attacker-2025"
  },
  {
   "event": "anthropic-opus-4-6-system-card-prompt-injection-2026",
   "note": "Claude Opus 4.6 with extended thinking, stronger Shade attacker: safeguards cut 200-attempt computer-use success from 78.6% to 57.1%."
  }
 ],
 "relations": [],
 "statusHistory": [
  {
   "status": "reported",
   "on": "2025-05-20",
   "why": "Google DeepMind: adversarial fine-tuning reduced but did not eliminate success.",
   "event": "deepmind-gemini-ipi-lessons-2025",
   "kind": "evidence"
  },
  {
   "status": "corroborated",
   "on": "2025-08-25",
   "why": "Anthropic reports residual success after mitigations in its browser agent.",
   "event": "anthropic-claude-in-chrome-pilot-pi-2025",
   "kind": "evidence"
  },
  {
   "status": "qualified",
   "on": "2026-02-05",
   "why": "Multi-attempt figures in Anthropic's system card are far higher than single-attempt rates.",
   "event": "anthropic-opus-4-6-system-card-prompt-injection-2026",
   "kind": "evidence"
  }
 ],
 "halfLifeDays": 270,
 "wouldChange": "Independent adaptive testing of current production defenses.",
 "fideQuestions": [],
 "methods": [
  "indirect-prompt-injection",
  "injection-classifiers",
  "instruction-priority-training"
 ],
 "review": "assistant-drafted",
 "addedOn": "2026-09-25"
}