{
 "license": "CC-BY-4.0",
 "attribution": "Fide AI, Agentic Cyber Explorer",
 "url": "https://agentic-cyber-explorer.pages.dev/events/actionguard-poisoned-skill-tool-call-authorization-2026/",
 "asOf": "2026-10-01",
 "id": "actionguard-poisoned-skill-tool-call-authorization-2026",
 "date": "2026-09-30",
 "datePrecision": "day",
 "title": "ActionGuard reports lower skill-injection success by authorizing tool calls against trusted user intent",
 "lane": "defense",
 "kind": "paper",
 "summary": "Korea University researchers propose ActionGuard, which checks skill-influenced tool calls immediately before execution against the trusted user request and runtime evidence. In an OpenClaw evaluation using SKILL-INJECT tasks, they report 8.65% overall attack success and 90.38% task success, averaged across reviewer models and injection types. The evaluation uses one framework, one target model and one benchmark; it does not establish performance against attackers adapting to the defense.",
 "whyItMatters": "It tests authorization at the execution boundary instead of treating skill instructions as permission, while measuring the resulting task-completion cost.",
 "actors": [
  "korea-university"
 ],
 "topics": [
  "prompt-injection",
  "agent-supply-chain",
  "access-controls",
  "monitoring-and-control"
 ],
 "atlas": [
  "tools",
  "untrusted-content",
  "monitor"
 ],
 "artifacts": [],
 "sources": [
  {
   "url": "https://arxiv.org/abs/2609.39450",
   "publisher": "arXiv",
   "title": "ActionGuard: Tool Call Authorization under Poisoned Skills",
   "date": "2026-09-30",
   "type": "primary",
   "accessed": "2026-10-01"
  }
 ],
 "keyFacts": [
  {
   "fact": "ActionGuard separates the agent’s planning context from an isolated reviewer’s authorization context. The reviewer uses the trusted request, an independently maintained skill profile, tool-call context and local script contents, rather than receiving the raw potentially poisoned skill as execution authority. Invalid or unavailable reviewer decisions deny execution.",
   "locator": "v1, Sections 4.1–4.3"
  },
  {
   "fact": "Evaluation uses 139 contextual and 180 obvious injection-task pairs in OpenClaw, five reviewer models, and three repetitions per condition. Comparisons include Dynamic Guardian, SkillGuard and no safeguard.",
   "locator": "v1, Sections 5.1–5.2; Tables 2–3"
  },
  {
   "fact": "Table 4 averages across reviewer models and injection types: ActionGuard 8.65% attack success and 90.38% task success; Dynamic Guardian 16.05% and 92.08%; SkillGuard 13.42% and 88.55%; no safeguard 29.05% and 93.46%. These are averaged rates, not individual-model results.",
   "locator": "v1, Table 4; Section 6.2"
  },
  {
   "fact": "The authors limit generalization to one agent framework, target model and benchmark. Blocking a call that combines benign and unauthorized actions can leave the task incomplete if the agent does not produce a safe alternative. Their abstract and conclusion give slightly different relative reductions against no safeguard; this record uses the absolute Table 4 rates.",
   "locator": "v1, Section 8, Conclusion; Abstract; Table 4"
  }
 ],
 "significance": 2,
 "fideQuestions": [
  "FID-074"
 ],
 "methods": [
  "ai-monitoring",
  "indirect-prompt-injection",
  "malicious-agent-extensions"
 ],
 "review": "assistant-drafted",
 "addedOn": "2026-10-01"
}