{
 "license": "CC-BY-4.0",
 "attribution": "Fide AI, Agentic Cyber Explorer",
 "url": "https://agentic-cyber-explorer.pages.dev/events/microsoft-cti-realm-benchmark-2026/",
 "asOf": "2026-09-26",
 "id": "microsoft-cti-realm-benchmark-2026",
 "date": "2026-03-13",
 "datePrecision": "day",
 "title": "Microsoft's CTI-REALM benchmark tests agents turning threat intel into validated detection rules",
 "lane": "defense",
 "kind": "benchmark",
 "summary": "CTI-REALM places agents in a tool-rich environment where they read threat intelligence reports, explore telemetry, iterate KQL queries and produce Sigma and KQL detection rules across Linux, AKS and Azure cloud scenarios. The paper's evaluation of 16 model configurations found Claude Opus 4.6 (High) best at 0.637, with cloud detection hardest; Microsoft's blog later added an early Claude Mythos Preview snapshot scoring 0.685.",
 "whyItMatters": "It measures an end-to-end detection engineering workflow, a core SOC task that most security benchmarks skip.",
 "actors": [
  "microsoft"
 ],
 "topics": [
  "soc-automation",
  "capability-evaluation"
 ],
 "atlas": [
  "eval-environment",
  "tools"
 ],
 "artifacts": [
  "cti-realm",
  "inspect",
  "claude-mythos",
  "claude-opus-4",
  "gpt-5-family"
 ],
 "sources": [
  {
   "url": "https://arxiv.org/abs/2603.13517",
   "publisher": "arXiv",
   "title": "CTI-REALM: Benchmark to Evaluate Agent Performance on Security Detection Rule Generation Capabilities",
   "date": "2026-03-13",
   "type": "primary",
   "accessed": "2026-09-25"
  },
  {
   "url": "https://www.microsoft.com/en-us/security/blog/2026/03/20/cti-realm-a-new-benchmark-for-end-to-end-detection-rule-generation-with-ai-agents/",
   "publisher": "Microsoft Security Blog",
   "title": "CTI-REALM: A new benchmark for end-to-end detection rule generation with AI agents",
   "date": "2026-03-20",
   "type": "primary",
   "accessed": "2026-09-25"
  }
 ],
 "keyFacts": [
  {
   "fact": "37 curated CTI reports; CTI-REALM-50 has 50 tasks across Linux, AKS and Azure cloud.",
   "locator": "Microsoft blog"
  },
  {
   "fact": "Across 16 frontier model configurations, Claude Opus 4.6 (High) achieved the highest reward (0.637), followed by Claude Opus 4.5 (0.624) and the GPT-5 family.",
   "locator": "arXiv abstract"
  },
  {
   "fact": "Scores fall from Linux (0.585) to AKS (0.517) to cloud (0.282); removing CTI-specific tools cut performance by up to 0.150.",
   "locator": "Microsoft blog, findings list"
  },
  {
   "fact": "The blog, updated after publication, reports an early Claude Mythos Preview snapshot at 0.685.",
   "locator": "Microsoft blog, results"
  }
 ],
 "significance": 3,
 "fideQuestions": [
  "FID-075",
  "FID-076"
 ],
 "methods": [
  "ctf-benchmarks"
 ],
 "review": "assistant-drafted",
 "addedOn": "2026-09-25"
}