{
 "license": "CC-BY-4.0",
 "attribution": "Fide AI, Agentic Cyber Explorer",
 "url": "https://agentic-cyber-explorer.pages.dev/events/openai-third-party-evaluation-playbook-2026/",
 "asOf": "2026-09-26",
 "id": "openai-third-party-evaluation-playbook-2026",
 "date": "2026-05-29",
 "datePrecision": "day",
 "title": "OpenAI publishes a playbook on harness choice and validity checks for third-party evaluations",
 "lane": "defense",
 "kind": "guidance",
 "summary": "OpenAI argues that agent evaluation reports must state which claim they test (capability ceiling, controlled comparison or safeguard robustness), describe harness, tools and budget, and show checks for reward hacking, refusals, contamination, broken problems and sandbagging. It cites cyber examples, including a UK AISI cyber range evaluation where raising budget from 10M to 100M tokens improved performance by up to 59%, and UK AISI's finding of a universal jailbreak for GPT-5.5 cyber safeguards using a custom harness.",
 "whyItMatters": "It is a lab's explicit statement that harness and compute choices can change cyber evaluation conclusions.",
 "actors": [
  "openai",
  "uk-aisi",
  "metr",
  "apollo-research"
 ],
 "topics": [
  "eval-validity",
  "capability-evaluation"
 ],
 "atlas": [
  "eval-environment"
 ],
 "artifacts": [
  "gpt-5-family"
 ],
 "sources": [
  {
   "url": "https://openai.com/index/trustworthy-third-party-evaluations-foundations/",
   "publisher": "OpenAI",
   "title": "A shared playbook for trustworthy third party evaluations",
   "date": "2026-05-29",
   "type": "primary",
   "accessed": "2026-09-25"
  }
 ],
 "keyFacts": [
  {
   "fact": "Cites UK AISI's cyber range evaluation: increasing token budget from 10M to 100M improved performance by up to 59%, still rising at the highest budget.",
   "locator": "Harness section"
  },
  {
   "fact": "Cites UK AISI's GPT-5.5 cyber evaluation, whose expert red team found a universal jailbreak eliciting violative cyber content, including in multi-turn agentic settings.",
   "locator": "Harness section"
  },
  {
   "fact": "OpenAI asks capability evaluators to use Codex as a common floor harness for OpenAI models.",
   "locator": "How we are supporting stronger evaluations"
  }
 ],
 "significance": 3,
 "fideQuestions": [
  "FID-075"
 ],
 "methods": [
  "adaptive-red-teaming",
  "compute-scaled-evaluation",
  "cyber-ranges",
  "evaluation-gaming"
 ],
 "review": "assistant-drafted",
 "addedOn": "2026-09-25"
}