{
 "license": "CC-BY-4.0",
 "attribution": "Fide AI, Agentic Cyber Explorer",
 "url": "https://agentic-cyber-explorer.pages.dev/findings/classifiers-cut-jailbreaks-not-to-zero/",
 "asOf": "2026-09-26",
 "id": "classifiers-cut-jailbreaks-not-to-zero",
 "claim": "Anthropic reports that classifier guards cut automated jailbreak success from 86% to 4.4%; a 2025 public demo still yielded one universal jailbreak, though none has been reported against its 2026 successor.",
 "evidenceKind": "reported",
 "scope": "Vendor-reported results for one lab's safeguards, built for chemical-weapons and other CBRN content rather than cyber misuse.",
 "topics": [
  "jailbreaks-and-safeguards"
 ],
 "atlas": [
  "model"
 ],
 "evidence": [
  {
   "event": "anthropic-constitutional-classifiers-2025",
   "note": "Automated jailbreak success on Claude 3.5 Sonnet fell from 86% to 4.4%; in the Feb 3-10, 2025 public demo one participant found a universal jailbreak."
  },
  {
   "event": "anthropic-next-gen-constitutional-classifiers-2026",
   "note": "Over 1,700 red-teaming hours and 198,000 attempts found no universal jailbreak; about 1% compute overhead."
  }
 ],
 "relations": [],
 "statusHistory": [
  {
   "status": "reported",
   "on": "2025-02-03",
   "why": "Anthropic Constitutional Classifiers.",
   "event": "anthropic-constitutional-classifiers-2025",
   "kind": "evidence"
  }
 ],
 "halfLifeDays": 365,
 "wouldChange": "Independent red-teaming of current cyber safeguards.",
 "fideQuestions": [],
 "methods": [
  "injection-classifiers",
  "jailbreaking"
 ],
 "review": "assistant-drafted",
 "addedOn": "2026-09-25"
}