{
 "license": "CC-BY-4.0",
 "attribution": "Fide AI, Agentic Cyber Explorer",
 "url": "https://agentic-cyber-explorer.pages.dev/findings/soc-agents-weak-on-realistic-benchmarks/",
 "asOf": "2026-09-26",
 "id": "soc-agents-weak-on-realistic-benchmarks",
 "claim": "LLM agents fall well short of reliable performance on realistic threat-investigation and threat-hunting benchmarks built from security logs.",
 "evidenceKind": "measured",
 "scope": "Benchmarks, several built by security vendors (Microsoft, Simbian); not measurements of deployed systems. CyberSOCEval is multiple-choice question answering, not an agentic task.",
 "topics": [
  "soc-automation",
  "autonomous-defense"
 ],
 "atlas": [],
 "evidence": [
  {
   "event": "microsoft-excytin-bench-2025",
   "note": "Best model reward 0.606 on log-investigation questions."
  },
  {
   "event": "meta-crowdstrike-cybersoceval-2025",
   "note": "Multiple-choice questions over malware sandbox and threat reports, not agentic tasks; models far from saturating."
  },
  {
   "event": "simbian-cyber-defense-benchmark-threat-hunting-2026",
   "note": "The best agent flagged 3.8% of malicious events in raw logs."
  }
 ],
 "relations": [
  {
   "type": "qualifies",
   "target": "assistants-speed-up-analysts",
   "note": "Gains appear in assisted work; autonomous performance on realistic tasks remains weak."
  }
 ],
 "statusHistory": [
  {
   "status": "reported",
   "on": "2025-07-14",
   "why": "ExCyTIn-Bench results.",
   "event": "microsoft-excytin-bench-2025",
   "kind": "evidence"
  },
  {
   "status": "corroborated",
   "on": "2025-09-24",
   "why": "CyberSOCEval finds similar limits.",
   "event": "meta-crowdstrike-cybersoceval-2025",
   "kind": "evidence"
  },
  {
   "status": "corroborated",
   "on": "2026-09-25",
   "why": "Correction: CyberSOCEval tests multiple-choice question answering, not agents on investigation or hunting. Corroboration rests on Simbian's Cyber Defense Benchmark, where the best of five models flagged 3.8% of malicious events in raw logs.",
   "event": "simbian-cyber-defense-benchmark-threat-hunting-2026",
   "kind": "correction"
  }
 ],
 "halfLifeDays": 365,
 "wouldChange": "Benchmarks showing agents reliably complete investigations.",
 "fideQuestions": [
  "FID-076"
 ],
 "methods": [
  "ctf-benchmarks"
 ],
 "review": "assistant-drafted",
 "addedOn": "2026-09-25"
}