{
 "license": "CC-BY-4.0",
 "attribution": "Fide AI, Agentic Cyber Explorer",
 "url": "https://agentic-cyber-explorer.pages.dev/questions/how-far-to-trust-cyber-capability-numbers/",
 "asOf": "2026-09-26",
 "id": "how-far-to-trust-cyber-capability-numbers",
 "order": 4,
 "question": "How far can measured AI cyber capability be trusted?",
 "topics": [
  "capability-evaluation",
  "eval-validity"
 ],
 "answers": [
  {
   "on": "2026-09-25",
   "short": "As a lower or conditional bound. Scores move substantially with token budget, evaluation pipeline, and benchmark contamination.",
   "body": "UK AISI reports that fixed, low token budgets understate frontier cyber capability and its rate of progress. A preprint audit finds pipeline choices alone can move cyber benchmark scores by tens of points, public CTF benchmarks can be contaminated, and counting a crash as exploitation overstates capability. Single capability numbers, including trend estimates, are best read as lower or conditional bounds.",
   "confidence": "moderate",
   "findings": [
    "fixed-budgets-understate-cyber-capability",
    "pipeline-choices-move-cyber-scores",
    "public-ctf-benchmarks-contaminated",
    "crash-is-not-exploitation"
   ],
   "methods": [
    "compute-scaled-evaluation",
    "ctf-benchmarks"
   ],
   "why": "First answer, drawn from the findings linked here."
  }
 ],
 "reviewedOn": "2026-09-25",
 "wouldChange": "Standardized reporting of budgets, pipelines, and contamination controls that makes results comparable across evaluators.",
 "review": "assistant-drafted"
}