{
 "license": "CC-BY-4.0",
 "attribution": "Fide AI, Agentic Cyber Explorer",
 "url": "https://agentic-cyber-explorer.pages.dev/findings/independent-tests-differ-from-self-reports/",
 "asOf": "2026-09-26",
 "id": "independent-tests-differ-from-self-reports",
 "claim": "Independent government testing can find a model weaker on agentic tasks than its developer's self-reported benchmarks suggest.",
 "evidenceKind": "measured",
 "scope": "One evaluation of one model on non-public benchmarks.",
 "topics": [
  "eval-validity",
  "capability-evaluation"
 ],
 "atlas": [],
 "evidence": [
  {
   "event": "caisi-deepseek-v4-pro-evaluation-2026",
   "note": "CAISI found weaker results on non-public agentic and reasoning benchmarks (ARC-AGI-2 semi-private, PortBench, CTF-Archive-Diamond) than DeepSeek self-reported."
  }
 ],
 "relations": [
  {
   "type": "supports",
   "target": "pipeline-choices-move-cyber-scores",
   "note": "Both show headline scores depend on evaluation choices; here the difference is which benchmarks are run, not how the same benchmark is run."
  }
 ],
 "statusHistory": [
  {
   "status": "reported",
   "on": "2026-05-01",
   "why": "CAISI evaluation of DeepSeek V4 Pro.",
   "event": "caisi-deepseek-v4-pro-evaluation-2026",
   "kind": "evidence"
  }
 ],
 "halfLifeDays": 270,
 "wouldChange": "Agreement between independent and self-reported results across several models.",
 "fideQuestions": [
  "FID-075"
 ],
 "methods": [
  "ctf-benchmarks"
 ],
 "review": "assistant-drafted",
 "addedOn": "2026-09-25"
}