{
 "license": "CC-BY-4.0",
 "attribution": "Fide AI, Agentic Cyber Explorer",
 "url": "https://agentic-cyber-explorer.pages.dev/events/dstl-evaluating-rl-cyber-defence-agents-2025/",
 "asOf": "2026-09-26",
 "id": "dstl-evaluating-rl-cyber-defence-agents-2025",
 "date": "2025-06-27",
 "datePrecision": "day",
 "title": "Paper proposes test and evaluation process with effectiveness metrics for RL cyber defence agents",
 "lane": "defense",
 "kind": "paper",
 "summary": "A paper in Applied AI Letters by QinetiQ researchers sets out a test and evaluation process for cyber defence agents covering performance, effectiveness, resilience and generalisability, and demonstrates its low-fidelity stage on CAGE Challenge 2 RL agents in CybORG. It introduces Measures of Effectiveness tailored to cyber defence alongside RL reward and tests agents under environment perturbations not seen in training.",
 "whyItMatters": "It proposes defence-specific effectiveness metrics and robustness tests to complement RL reward when judging whether a defensive agent can be trusted.",
 "actors": [],
 "topics": [
  "autonomous-defense",
  "eval-validity"
 ],
 "atlas": [
  "eval-environment"
 ],
 "artifacts": [],
 "sources": [
  {
   "url": "https://doi.org/10.1002/ail2.125",
   "publisher": "Applied AI Letters (Wiley)",
   "title": "Evaluating Reinforcement Learning Agents for Autonomous Cyber Defence",
   "date": "2025-06-27",
   "type": "primary",
   "accessed": "2026-09-26"
  },
  {
   "url": "https://research-information.bris.ac.uk/en/publications/d98a4af3-e7c8-4cfe-98e8-368233cf9b6a",
   "publisher": "University of Bristol",
   "title": "Evaluating Reinforcement Learning Agents for Autonomous Cyber Defence (repository record)",
   "type": "primary",
   "accessed": "2026-09-26"
  },
  {
   "url": "https://api.crossref.org/works/10.1002/ail2.125",
   "publisher": "Crossref (publisher-deposited metadata)",
   "title": "Evaluating Reinforcement Learning Agents for Autonomous Cyber Defence",
   "date": "2025-06-27",
   "type": "secondary",
   "accessed": "2026-09-25"
  }
 ],
 "keyFacts": [
  {
   "fact": "The process evaluates performance, effectiveness, resilience and generalisability in low- and high-fidelity environments; the paper demonstrates the low-fidelity stage on CAGE Challenge 2 agents.",
   "locator": "Abstract"
  },
  {
   "fact": "Agents are evaluated against perturbed conditions to test robustness to scenarios not seen during training.",
   "locator": "Abstract"
  }
 ],
 "significance": 2,
 "fideQuestions": [
  "FID-075",
  "FID-076"
 ],
 "methods": [
  "cyber-ranges"
 ],
 "review": "assistant-drafted",
 "addedOn": "2026-09-25"
}