{
 "license": "CC-BY-4.0",
 "attribution": "Fide AI, Agentic Cyber Explorer",
 "url": "https://agentic-cyber-explorer.pages.dev/findings/fixed-budgets-understate-cyber-capability/",
 "asOf": "2026-09-26",
 "id": "fixed-budgets-understate-cyber-capability",
 "claim": "Cyber capability measured at fixed, low token budgets understates what frontier models can do and how fast they are improving.",
 "evidenceKind": "measured",
 "scope": "UK AISI measurements on its task suite; the size of the effect varies by model and task.",
 "topics": [
  "eval-validity",
  "capability-evaluation"
 ],
 "atlas": [],
 "evidence": [
  {
   "event": "openai-third-party-evaluation-playbook-2026"
  },
  {
   "event": "uk-aisi-test-time-compute-agent-evals-2026",
   "note": "Raising the budget from 2.5M to 50M tokens moved one model's cyber time horizon from about 40 minutes to about 4 hours."
  },
  {
   "event": "uk-aisi-sandboxescapebench-2026"
  },
  {
   "event": "aisi-cyber-time-horizons-2026",
   "note": "UK AISI's May post already said the 2.5M-token cap understates frontier capability."
  }
 ],
 "relations": [
  {
   "type": "qualifies",
   "target": "cyber-time-horizons-doubling",
   "note": "The doubling estimate was measured at a fixed low budget."
  }
 ],
 "statusHistory": [
  {
   "status": "reported",
   "on": "2026-05-29",
   "why": "OpenAI's evaluation playbook warns that unreported budgets understate capability.",
   "event": "openai-third-party-evaluation-playbook-2026",
   "kind": "evidence"
  },
  {
   "status": "corroborated",
   "on": "2026-07-02",
   "why": "UK AISI measures the effect directly.",
   "event": "uk-aisi-test-time-compute-agent-evals-2026",
   "kind": "evidence"
  },
  {
   "status": "reported",
   "on": "2026-09-25",
   "why": "Correction: the OpenAI playbook cites UK AISI's own measurements, so all evidence comes from one evaluator.",
   "event": "uk-aisi-test-time-compute-agent-evals-2026",
   "kind": "correction"
  }
 ],
 "halfLifeDays": 270,
 "wouldChange": "Evidence that capability plateaus at budgets in common use.",
 "fideQuestions": [
  "FID-075"
 ],
 "methods": [
  "compute-scaled-evaluation",
  "ctf-benchmarks"
 ],
 "review": "assistant-drafted",
 "addedOn": "2026-09-25"
}