{
 "license": "CC-BY-4.0",
 "attribution": "Fide AI, Agentic Cyber Explorer",
 "url": "https://agentic-cyber-explorer.pages.dev/events/uk-aisi-test-time-compute-agent-evals-2026/",
 "asOf": "2026-09-26",
 "id": "uk-aisi-test-time-compute-agent-evals-2026",
 "date": "2026-07-02",
 "datePrecision": "day",
 "title": "UK AISI finds agent evaluations understate cyber capability without accounting for test-time compute",
 "lane": "defense",
 "kind": "eval-report",
 "summary": "UK AISI's Science of Evaluation team measured how agent success changes with token budget across software, academic and cyber tasks. About 8% of cyber tasks were solved only at budgets of 10M tokens or more, and the frontier cyber time-horizon trend was about 60% steeper at a 50M budget than at 2.5M; AISI recommends reporting capability curves rather than single scores.",
 "whyItMatters": "Single-budget cyber evaluation scores can miss capability that appears at higher, attacker-affordable compute.",
 "actors": [
  "uk-aisi"
 ],
 "topics": [
  "eval-validity",
  "capability-evaluation"
 ],
 "atlas": [
  "eval-environment"
 ],
 "artifacts": [],
 "sources": [
  {
   "url": "https://www.aisi.gov.uk/blog/more-compute-more-capability-why-ai-agent-evals-need-to-account-for-test-time-compute",
   "publisher": "UK AI Security Institute",
   "title": "More compute, more capability: Why AI agent evaluations need to account for test-time compute",
   "date": "2026-07-02",
   "type": "primary",
   "accessed": "2026-09-25"
  }
 ],
 "keyFacts": [
  {
   "fact": "About 8% of cyber tasks were solved only at 10M+ token budgets, some requiring up to 50M tokens.",
   "locator": "Our findings"
  },
  {
   "fact": "Cyber time horizons doubled every 4.7 months at a 2.5M budget; the trend is about 60% steeper at 50M; one frontier model's horizon rose from about 40 minutes (2.5M) to about 4 hours (50M).",
   "locator": "Our findings"
  },
  {
   "fact": "Human task time predicted agent compute need via a power law (exponent about 0.7-1.0) across 211 software and 78 cyber tasks.",
   "locator": "Our findings"
  }
 ],
 "significance": 4,
 "fideQuestions": [
  "FID-075"
 ],
 "methods": [
  "compute-scaled-evaluation",
  "ctf-benchmarks"
 ],
 "review": "assistant-drafted",
 "addedOn": "2026-09-25"
}