{
 "license": "CC-BY-4.0",
 "attribution": "Fide AI, Agentic Cyber Explorer",
 "url": "https://agentic-cyber-explorer.pages.dev/events/trustprobe-skill-agent-trust-failures-2026/",
 "asOf": "2026-10-01",
 "id": "trustprobe-skill-agent-trust-failures-2026",
 "date": "2026-09-30",
 "datePrecision": "day",
 "title": "TrustProbe reports 104 skill-mediated trust failures across eleven agents under permissive test settings",
 "lane": "attack",
 "kind": "paper",
 "summary": "Researchers from the Chinese Academy of Sciences and Worcester Polytechnic Institute introduce TrustProbe to trace installed skill content into security-sensitive operations and verify observable effects. Using DeepSeek-V4-Flash across eleven open-source agents, they report 104 verified source-to-sink vulnerabilities in permissive non-interactive configurations, with a subset remaining exploitable under stricter approval settings. Their real-skill experiment measures exposure to vulnerable execution paths, rather than the prevalence of malicious skills or attacks on real users.",
 "whyItMatters": "It distinguishes ordinary skill-path exposure from verified harmful behavior and tests how installation and approval mechanisms affect delegated authority.",
 "actors": [
  "iie-cas",
  "worcester-polytechnic-institute",
  "university-of-chinese-academy-of-sciences"
 ],
 "topics": [
  "agent-supply-chain",
  "tool-and-mcp-security",
  "access-controls",
  "prompt-injection"
 ],
 "atlas": [
  "supply-chain",
  "tools",
  "human-approver"
 ],
 "artifacts": [
  "deepseek"
 ],
 "sources": [
  {
   "url": "https://arxiv.org/abs/2609.39065",
   "publisher": "arXiv",
   "title": "Can Agents Trust Their Skills? Uncovering Unsafe Chains of Trust in Skill-Based LLM Agents",
   "date": "2026-09-30",
   "type": "primary",
   "accessed": "2026-10-01"
  }
 ],
 "keyFacts": [
  {
   "fact": "The main campaign runs eleven skill-supporting open-source agents in fresh sessions and isolated workspaces with permissive non-interactive execution settings. DeepSeek-V4-Flash supplies both target-agent inference and TrustProbe generation, scoring and mutation; this is not an eleven-model comparison.",
   "locator": "v1, Sections 5.1–5.2"
  },
  {
   "fact": "The authors report 104 verified taint-style vulnerabilities, counted once per audited source-to-sink path when attacker-controlled flow and observable harm are both confirmed. Under the strictest usable non-interactive approval settings, 31 of 89 applicable vulnerabilities across eight agents remain exploitable.",
   "locator": "v1, Section 5.2; Table 1; Appendices H–I"
  },
  {
   "fact": "Direct-prompt replay reproduces 33 of the 104 discovery-confirmed vulnerabilities (31.7%). This compares reproduction of previously discovered skill cases with prompt delivery, rather than equal-budget discovery campaigns.",
   "locator": "v1, Section 5.3; Table 2"
  },
  {
   "fact": "Among 2,963 compatible skill-agent runs using 633 real skill files, 743 (25.1%) trigger an audited vulnerable path. This exposure rate does not label the original skills malicious. The authors validate complete attacks after modifying 15 selected triggering skills, in isolated workspaces using controlled data and endpoints.",
   "locator": "v1, Section 5.4; Ethics statement"
  }
 ],
 "significance": 3,
 "fideQuestions": [
  "FID-076"
 ],
 "methods": [
  "malicious-agent-extensions",
  "indirect-prompt-injection",
  "human-approval"
 ],
 "review": "assistant-drafted",
 "addedOn": "2026-10-01"
}