{
 "license": "CC-BY-4.0",
 "attribution": "Fide AI, Agentic Cyber Explorer",
 "url": "https://agentic-cyber-explorer.pages.dev/events/aixcc-sok-competition-lessons-2026/",
 "asOf": "2026-09-26",
 "id": "aixcc-sok-competition-lessons-2026",
 "date": "2026-02-07",
 "datePrecision": "day",
 "title": "AIxCC SoK finds stability decided results and many validated AI patches were still semantically wrong",
 "lane": "defense",
 "kind": "paper",
 "summary": "A systematization-of-knowledge paper by organizers and competitors analyzes AIxCC's design, the seven finalist architectures and results beyond the scoreboard. It reports that system stability and accuracy penalties decided rankings, that LLM-based systems found vulnerabilities a fuzzing baseline missed, and that among patches passing all automatic validation, manual review found semantic errors in 38-46% from baseline agents; the top two systems had 83.8% and 79.2% competition-scored patch accuracy.",
 "whyItMatters": "It gives a detailed account, beyond the scoreboard, of what autonomous cyber reasoning systems achieved and where their patches failed.",
 "actors": [
  "georgia-tech",
  "texas-am-university",
  "darpa",
  "team-atlanta",
  "trail-of-bits",
  "theori"
 ],
 "topics": [
  "vulnerability-repair",
  "vulnerability-discovery",
  "eval-validity"
 ],
 "atlas": [
  "eval-environment"
 ],
 "artifacts": [
  "aixcc",
  "buttercup",
  "oss-crs"
 ],
 "sources": [
  {
   "url": "https://arxiv.org/abs/2602.07666",
   "publisher": "arXiv",
   "title": "SoK: DARPA's AI Cyber Challenge (AIxCC): Competition Design, Architectures, and Lessons Learned",
   "date": "2026-02-07",
   "type": "primary",
   "accessed": "2026-09-25"
  },
  {
   "url": "https://arxiv.org/html/2602.07666",
   "publisher": "arXiv",
   "title": "SoK: DARPA's AI Cyber Challenge (AIxCC) (HTML, v5)",
   "date": "2026-08-02",
   "type": "primary",
   "accessed": "2026-09-25"
  },
  {
   "url": "https://occia.github.io/aixcc-sok-webpage/",
   "publisher": "AIxCC SoK companion site",
   "title": "AIxCC SoK companion site",
   "type": "primary",
   "accessed": "2026-09-25"
  }
 ],
 "keyFacts": [
  {
   "fact": "Team Atlanta scored 392.8 points vs 219.4 for second place; Theori's pre-penalty score exceeded Trail of Bits but a -16.3 accuracy penalty dropped it to third.",
   "locator": "Section 7.1"
  },
  {
   "fact": "A parallel-fuzzing baseline solved 34 of 63 challenge vulnerabilities (54%) but only 4 of 23 Java ones; six CRSs found 7-16 vulnerabilities the baseline missed.",
   "locator": "Section 7.3"
  },
  {
   "fact": "Competition-scored patch accuracy was 83.8% (Team Atlanta) and 79.2% (Trail of Bits); manual review of baseline-agent patches that passed all automatic checks found 37.7% (Claude Code) and 45.6% (MultiRetrieval) semantically incorrect.",
   "locator": "Table 10; Section 7.4 (KF 6-7)"
  },
  {
   "fact": "About 94% of LLM spending went to Anthropic and OpenAI models; no team exhausted its $50K LLM credit or $85K compute budget.",
   "locator": "Section 7.5, Table 8"
  },
  {
   "fact": "Companion site lists raw competition data (submission logs, traces, scoring breakdowns) as pending release.",
   "locator": "Companion site, artifacts list"
  },
  {
   "fact": "Accepted to USENIX Security 2026; v1 7 Feb 2026, v5 2 Aug 2026.",
   "locator": "arXiv listing"
  }
 ],
 "significance": 4,
 "fideQuestions": [
  "FID-075",
  "FID-088"
 ],
 "methods": [
  "adaptive-red-teaming",
  "ai-vulnerability-discovery",
  "automated-patching",
  "compute-scaled-evaluation",
  "patch-verification"
 ],
 "review": "assistant-drafted",
 "addedOn": "2026-09-25"
}