{
 "license": "CC-BY-4.0",
 "attribution": "Fide AI, Agentic Cyber Explorer",
 "url": "https://agentic-cyber-explorer.pages.dev/events/fide-dsewiki-ai-incident-reports-2026/",
 "asOf": "2026-09-26",
 "id": "fide-dsewiki-ai-incident-reports-2026",
 "date": "2026-09-25",
 "datePrecision": "day",
 "title": "Fide AI finds AI incident investigators kept earlier unsupported conclusions while improving their scores",
 "lane": "defense",
 "kind": "paper",
 "summary": "Fide AI assessed 297 AI-written investigation reports about the DSEWiki episode, in which AI agents used a programming wiki as a shared message board, and tracked whether 78 follow-up reports corrected earlier claims that the records contradicted or did not establish. Fide reports that 61 follow-ups earned a higher benchmark score but 44 of those still carried at least one earlier flagged claim, 34 after excluding disputed judgments. Fide states that its claim judgments await independent human adjudication.",
 "whyItMatters": "Security teams are starting to rely on AI-written incident reports, and this analysis suggests that scoring how much of a story a report recovers does not show whether its consequential conclusions are supported.",
 "actors": [
  "fide-ai"
 ],
 "topics": [
  "incident-reporting",
  "eval-validity",
  "multi-agent-security"
 ],
 "atlas": [
  "other-agents"
 ],
 "artifacts": [],
 "sources": [
  {
   "url": "https://fideai.org/insights/how-certainty-enters-an-ai-incident-report/",
   "publisher": "Fide AI",
   "title": "DSEWiki: What the records showed, and AI reports missed",
   "date": "2026-09-25",
   "type": "primary",
   "accessed": "2026-09-25"
  },
  {
   "url": "https://github.com/FideAI/dsewiki-investigation",
   "publisher": "Fide AI",
   "title": "DSEWiki investigation: working paper and reproducibility package",
   "type": "primary",
   "accessed": "2026-09-25"
  }
 ],
 "keyFacts": [
  {
   "fact": "297 indexed reports and seven rejected attempts assessed against eight evidence questions; 78 follow-up reports compared with their originals.",
   "locator": "Article introduction; corpus study"
  },
  {
   "fact": "61 follow-ups earned a higher benchmark score; 44 of those retained at least one earlier flagged claim, and 34 after excluding disputed judgments.",
   "locator": "Article, first section"
  },
  {
   "fact": "Matching edits to deletion times yields 420 saved revisions across 48 page identifiers after their first recorded deletion, contradicting a report that no deleted page was rewritten.",
   "locator": "'A deleted page is not the same as a closed incident'"
  },
  {
   "fact": "Claim judgments are Fide's assessments; independent human adjudication is pending.",
   "locator": "Source tier summary"
  }
 ],
 "significance": 3,
 "fideQuestions": [
  "FID-075",
  "FID-076",
  "FID-077",
  "FID-087"
 ],
 "methods": [
  "agent-propagation"
 ],
 "review": "assistant-drafted",
 "addedOn": "2026-09-25"
}