{
 "license": "CC-BY-4.0",
 "attribution": "Fide AI, Agentic Cyber Explorer",
 "url": "https://agentic-cyber-explorer.pages.dev/events/benchmark-scores-pipeline-dependent-cyber-2026/",
 "asOf": "2026-09-26",
 "id": "benchmark-scores-pipeline-dependent-cyber-2026",
 "date": "2026-09-08",
 "datePrecision": "day",
 "title": "Audit finds cybersecurity LLM benchmark scores swing over 80 points with evaluation pipeline choices",
 "lane": "defense",
 "kind": "paper",
 "summary": "Berriche, Shalby, Alhanahnah and Boshmaf audit eight cybersecurity benchmarks across 10 proprietary, open-weight and security-specialized LLMs. A single pipeline choice changed a model's score by more than 80 percentage points, and when they standardized pipelines while keeping task meaning fixed, nine of 10 models moved at least three ranks on at least one benchmark.",
 "whyItMatters": "Published cyber benchmark rankings may reflect harness and parsing choices as much as model capability.",
 "actors": [],
 "topics": [
  "eval-validity",
  "capability-evaluation"
 ],
 "atlas": [
  "eval-environment"
 ],
 "artifacts": [
  "gpt-5-family"
 ],
 "sources": [
  {
   "url": "https://arxiv.org/abs/2609.08765",
   "publisher": "arXiv",
   "title": "Benchmark Scores Are Pipeline-Dependent: A Reliability Audit of Cybersecurity LLM Benchmarks",
   "date": "2026-09-08",
   "type": "primary",
   "accessed": "2026-09-25"
  }
 ],
 "keyFacts": [
  {
   "fact": "Audit of eight cybersecurity benchmarks (including CyberMetric, SecEval, CTI-Bench and AthenaBench), 48,662 questions across 23 tasks, against 10 LLMs; 15 recurring pipeline failure modes identified.",
   "locator": "Abstract; Introduction; Sections 3-4"
  },
  {
   "fact": "A single pipeline choice can change a model's score by more than 80 percentage points.",
   "locator": "Abstract; Introduction"
  },
  {
   "fact": "Under a harness that standardizes pipeline choices while preserving task semantics, 9 of 10 models shift at least three ranks on at least one benchmark; GPT-5.4 is the exception.",
   "locator": "Abstract; Table 5"
  }
 ],
 "significance": 4,
 "fideQuestions": [
  "FID-075"
 ],
 "methods": [
  "ctf-benchmarks"
 ],
 "review": "assistant-drafted",
 "addedOn": "2026-09-25"
}