{
 "license": "CC-BY-4.0",
 "attribution": "Fide AI, Agentic Cyber Explorer",
 "url": "https://agentic-cyber-explorer.pages.dev/events/us-aisi-agent-hijacking-evaluations-2025/",
 "asOf": "2026-09-26",
 "id": "us-aisi-agent-hijacking-evaluations-2025",
 "date": "2025-01-17",
 "datePrecision": "day",
 "title": "US AISI (later CAISI) shows red-team attacks and repeated attempts raise agent hijacking rates on AgentDojo",
 "lane": "defense",
 "kind": "eval-report",
 "summary": "NIST's AI safety institute technical staff (renamed the Center for AI Standards and Innovation in June 2025) extended AgentDojo and red-teamed agents built on the upgraded Claude 3.5 Sonnet. On held-out Workspace tasks, attack success rose from 11% for the strongest baseline attack to 81% for the strongest newly developed attack, and across five injection tasks from 57% to 80% when each attack was tried 25 times. The team released an Inspect-based AgentDojo port and ran the red teaming with the UK AI Security Institute.",
 "whyItMatters": "A government evaluator showed that agent-hijacking scores depend heavily on attack novelty and attempt count, not only on the model.",
 "actors": [
  "nist",
  "us-caisi",
  "uk-aisi",
  "anthropic"
 ],
 "topics": [
  "prompt-injection",
  "eval-validity"
 ],
 "atlas": [
  "untrusted-content",
  "tools",
  "eval-environment"
 ],
 "artifacts": [
  "agentdojo",
  "claude-sonnet"
 ],
 "sources": [
  {
   "url": "https://www.nist.gov/news-events/news/2025/01/technical-blog-strengthening-ai-agent-hijacking-evaluations",
   "publisher": "NIST",
   "title": "Technical Blog: Strengthening AI Agent Hijacking Evaluations",
   "date": "2025-01-17",
   "type": "primary",
   "accessed": "2026-09-25"
  }
 ],
 "keyFacts": [
  {
   "fact": "Claude 3.5 Sonnet (Oct 2024), held-out Workspace tasks: 11% ASR for the strongest AgentDojo baseline attack vs 81% for the strongest new red-team attack.",
   "locator": "Insight #2 section"
  },
  {
   "fact": "Across five injection tasks, average ASR was 57% at one attempt and 80% with 25 attempts per attack.",
   "locator": "Insight #3 and #4 sections"
  }
 ],
 "significance": 4,
 "fideQuestions": [
  "FID-075"
 ],
 "methods": [
  "adaptive-red-teaming"
 ],
 "review": "assistant-drafted",
 "addedOn": "2026-09-25"
}