{
 "license": "CC-BY-4.0",
 "attribution": "Fide AI, Agentic Cyber Explorer",
 "url": "https://agentic-cyber-explorer.pages.dev/events/aptinvestbench-cross-telemetry-investigation-2026/",
 "asOf": "2026-10-01",
 "id": "aptinvestbench-cross-telemetry-investigation-2026",
 "date": "2026-09-30",
 "datePrecision": "day",
 "title": "APTInvestBench finds autonomous investigators lose citation support when telemetry changes",
 "lane": "defense",
 "kind": "benchmark",
 "summary": "Zhongguancun Laboratory researchers introduce APTInvestBench, built from report-informed attack reconstructions under varied log-collection conditions. In their ten-scenario comparison of eleven models, agents acquire sufficient evidence for 44.3% of recoverable attack actions on average, but their formal citations support 25.0%; similar aggregate scores conceal losses in which actions remain supported. These are controlled benchmark investigations, not measurements of deployed SOC performance.",
 "whyItMatters": "It distinguishes missing telemetry from an agent failing to find evidence or preserve it in its report, making aggregate investigation scores easier to interpret.",
 "actors": [
  "zhongguancun-laboratory"
 ],
 "topics": [
  "autonomous-defense",
  "soc-automation",
  "capability-evaluation",
  "eval-validity"
 ],
 "atlas": [
  "eval-environment"
 ],
 "artifacts": [],
 "sources": [
  {
   "url": "https://arxiv.org/abs/2609.38954",
   "publisher": "arXiv",
   "title": "APTInvestBench: Evaluating Autonomous APT Investigation under Varying Telemetry",
   "date": "2026-09-30",
   "type": "primary",
   "accessed": "2026-10-01"
  }
 ],
 "keyFacts": [
  {
   "fact": "The resource contains 370 cases across seven SOC-inspired telemetry conditions, derived from 56 report-informed attack reconstructions and approximately 16.4 million log records. The main evaluation concentrates on ten of the 56 scenarios.",
   "locator": "v1, Abstract; Section 5, Experimental setup"
  },
  {
   "fact": "Eleven-model comparisons use Codex; framework comparisons hold MiniMax-M3 fixed across Codex, Claude Code, OpenCode and Ref-ReAct. Across the seven telemetry conditions, models acquire sufficient evidence for 44.3% of recoverable actions on average, while formal citations support 25.0%.",
   "locator": "v1, Sections 5 and 5.1; Figure 3"
  },
  {
   "fact": "On actions recoverable in both Full and endpoint-only telemetry, pooled citation coverage changes from 27.5% to 25.9%. The authors report that 64.5% of actions supported in Full reports remain supported in endpoint-only reports; newly covered actions partly offset lost support. These comparisons allow the available supporting records to differ.",
   "locator": "v1, Section 5.2; Figure 4"
  },
  {
   "fact": "A stricter comparison holds registered candidate records and sufficient-evidence sets unchanged. Investigation results still differ across telemetry views; all four evaluated frameworks have lower citation coverage under endpoint-only telemetry.",
   "locator": "v1, Section 5.3; Figure 6"
  }
 ],
 "significance": 3,
 "fideQuestions": [
  "FID-075",
  "FID-076"
 ],
 "methods": [],
 "review": "assistant-drafted",
 "addedOn": "2026-10-01"
}