{
 "license": "CC-BY-4.0",
 "attribution": "Fide AI, Agentic Cyber Explorer",
 "url": "https://agentic-cyber-explorer.pages.dev/events/microsoft-excytin-bench-2025/",
 "asOf": "2026-09-26",
 "id": "microsoft-excytin-bench-2025",
 "date": "2025-07-14",
 "datePrecision": "day",
 "title": "Microsoft's ExCyTIn-Bench evaluates LLM agents on multi-step threat investigation over Sentinel logs",
 "lane": "defense",
 "kind": "benchmark",
 "summary": "ExCyTIn-Bench builds threat-investigation questions from graphs of security logs collected in a controlled Azure tenant with simulated multi-step attacks, and asks agents to query the logs to answer them. In the July 2025 version the best model (o4-mini) reached a reward of 0.368; in the May 2026 revision, accepted at ICML 2026, the best (Claude Opus 4.5) reached 0.606, which the authors say leaves substantial headroom.",
 "whyItMatters": "It is an open benchmark for the investigative, log-querying work of SOC analysts rather than multiple-choice knowledge.",
 "actors": [
  "microsoft"
 ],
 "topics": [
  "soc-automation",
  "capability-evaluation"
 ],
 "atlas": [
  "eval-environment",
  "tools"
 ],
 "artifacts": [
  "excytin-bench",
  "claude-opus-4",
  "openai-o-series"
 ],
 "sources": [
  {
   "url": "https://arxiv.org/abs/2507.14201",
   "publisher": "arXiv",
   "title": "ExCyTIn-Bench: Evaluating LLM agents on Cyber Threat Investigation",
   "date": "2025-07-14",
   "type": "primary",
   "accessed": "2026-09-25"
  },
  {
   "url": "https://github.com/microsoft/SecRL",
   "publisher": "GitHub (Microsoft)",
   "title": "SecRL: Benchmarking LLM agents on Cyber Threat Investigation",
   "type": "primary",
   "accessed": "2026-09-25"
  }
 ],
 "keyFacts": [
  {
   "fact": "v1 (July 2025): 8 simulated multi-step attacks in a controlled Azure tenant, 57 log tables from Microsoft Sentinel and related services, 589 generated questions; average reward 0.249 across evaluated models, best 0.368 (o4-mini).",
   "locator": "arXiv v1 abstract; Table 2"
  },
  {
   "fact": "v3 (May 2026, ICML 2026 version): 7,542 generated questions from the same 57 log tables; best reward 0.606 (Claude Opus 4.5).",
   "locator": "arXiv v3 abstract; Section 1; Table 2"
  }
 ],
 "significance": 3,
 "fideQuestions": [
  "FID-075",
  "FID-076"
 ],
 "methods": [],
 "review": "assistant-drafted",
 "addedOn": "2026-09-25"
}