{
 "license": "CC-BY-4.0",
 "attribution": "Fide AI, Agentic Cyber Explorer",
 "url": "https://agentic-cyber-explorer.pages.dev/events/microsoft-phishing-triage-agent-rct-2025/",
 "asOf": "2026-09-26",
 "id": "microsoft-phishing-triage-agent-rct-2025",
 "date": "2025-11-17",
 "datePrecision": "day",
 "title": "Microsoft RCT finds phishing triage agent raised analysts' true positives per minute up to 6.5x",
 "lane": "defense",
 "kind": "paper",
 "summary": "Microsoft reports a randomized controlled trial of its own Security Copilot Phishing Triage Agent. In the trial, 167 external security analysts each triaged a 25-email queue drawn from a curated corpus of emails reported by Microsoft employees. In the scenario where the agent classified every corpus email correctly, analysts with the agent found 6.5 times as many true positives per minute as the control group and scored 77% higher on F1; with the agent's accuracy set to 80% and a 20% malicious rate, the productivity gain fell to 3.1 times. Analysts with the agent spent 53% more time on malicious emails and did not simply confirm its malicious verdicts, but they were more likely to accept its benign verdicts, including planted false negatives.",
 "whyItMatters": "It is one of the few randomized measurements of a commercial SOC triage agent's effect on analysts, and it reports automation bias toward the agent's benign verdicts alongside the productivity gains. It is a vendor study of its own product in a controlled task, not live operations.",
 "actors": [
  "microsoft"
 ],
 "topics": [
  "soc-automation",
  "autonomous-defense"
 ],
 "atlas": [
  "human-approver"
 ],
 "artifacts": [
  "security-copilot"
 ],
 "sources": [
  {
   "url": "https://arxiv.org/abs/2511.13860",
   "publisher": "arXiv",
   "title": "Randomized Controlled Trials for Phishing Triage Agent",
   "date": "2025-11-17",
   "type": "primary",
   "accessed": "2026-09-25"
  },
  {
   "url": "https://cdn-dynmedia-1.microsoft.com/is/content/microsoftcorp/microsoft/bade/documents/products-and-services/en-us/security/randomized-controlled-trial-for-phishing-triage-agent-accessible.pdf",
   "publisher": "Microsoft",
   "title": "Randomized Controlled Trial for Phishing Triage Agent (PDF)",
   "date": "2025-11",
   "type": "primary",
   "accessed": "2026-09-25"
  }
 ],
 "keyFacts": [
  {
   "fact": "True positives per analyst minute rose 6.53x for the Aware group vs Control under the corpus ground truth (agent precision and recall 100% on the corpus; benign-resolution protocol applied), and 3.07x under the pessimistic counterfactual (80% agent precision and recall, 20% malicious rate).",
   "locator": "Section 3.1, Table 1"
  },
  {
   "fact": "Queue prioritization (benign-resolution) accounted for 78-87% of the productivity gain across scenarios; the agent's verdicts and explanations accounted for 13-22%.",
   "locator": "Section 3.1, Table 1"
  },
  {
   "fact": "F1 score rose from 0.53 (Control) to 0.93 (+77%) under the corpus ground truth, and to 0.77 (+48%) under the 80%-accuracy counterfactual.",
   "locator": "Section 3.2, Table 2"
  },
  {
   "fact": "Aware analysts spent 53% more time per malicious email and confirmed agent true positives at a similar rate to Control (88% vs 82%, not significant), but confirmed agent benign verdicts 21 pp more often and missed synthetic false negatives 29 pp more often (46% vs 17%).",
   "locator": "Section 3.3, Tables 3-4"
  },
  {
   "fact": "Design: 167 external analysts randomized into Control (52), Aware (56) and Blind (59) groups; each triaged 25 emails sampled from a 93-email corpus of Microsoft-employee reports (11 malicious), with synthetic false positives and negatives added.",
   "locator": "Section 2; Appendix A, Table 5"
  }
 ],
 "significance": 3,
 "fideQuestions": [
  "FID-076"
 ],
 "methods": [],
 "review": "assistant-drafted",
 "addedOn": "2026-09-25"
}