{
 "license": "CC-BY-4.0",
 "attribution": "Fide AI, Agentic Cyber Explorer",
 "url": "https://agentic-cyber-explorer.pages.dev/events/anthropic-next-gen-constitutional-classifiers-2026/",
 "asOf": "2026-09-26",
 "id": "anthropic-next-gen-constitutional-classifiers-2026",
 "date": "2026-01-09",
 "datePrecision": "day",
 "title": "Anthropic's next-generation Constitutional Classifiers cut overhead to about 1% using probe cascades",
 "lane": "defense",
 "kind": "paper",
 "summary": "Anthropic describes Constitutional Classifiers++, a cascade in which a cheap linear probe on model activations screens all traffic and escalates flagged exchanges to a probe-classifier ensemble. It reports roughly 1% added compute if applied to Claude Opus 4.0 traffic (the first generation added 23.7%) and a 0.05% refusal rate on harmless queries over one month of Claude Sonnet 4.5 traffic. Red-teamers found no universal jailbreak in over 1,700 hours.",
 "whyItMatters": "Cheaper classifier guards make it more practical to run misuse safeguards on all traffic. The reported results are for CBRN safeguards; whether they carry over to cyber misuse is not shown.",
 "actors": [
  "anthropic"
 ],
 "topics": [
  "jailbreaks-and-safeguards"
 ],
 "atlas": [
  "model",
  "monitor"
 ],
 "artifacts": [
  "constitutional-classifiers",
  "claude-opus-4",
  "claude-sonnet"
 ],
 "sources": [
  {
   "url": "https://www.anthropic.com/research/next-generation-constitutional-classifiers",
   "publisher": "Anthropic",
   "title": "Next-generation Constitutional Classifiers: More efficient protection against universal jailbreaks",
   "date": "2026-01-09",
   "type": "primary",
   "accessed": "2026-09-25"
  }
 ],
 "keyFacts": [
  {
   "fact": "About 1% additional compute if applied to Claude Opus 4.0 traffic, versus a 23.7% compute increase for the first-generation classifiers.",
   "locator": "Introduction; Conclusions and further research"
  },
  {
   "fact": "0.05% refusal rate on harmless queries over one month of deployment on Claude Sonnet 4.5 traffic, 87% lower than the original classifier system.",
   "locator": "Conclusions and further research"
  },
  {
   "fact": "Over 1,700 cumulative red-teaming hours across 198,000 attempts found one high-risk vulnerability and no universal jailbreak.",
   "locator": "Conclusions and further research"
  }
 ],
 "significance": 3,
 "fideQuestions": [],
 "methods": [
  "ai-vulnerability-discovery",
  "injection-classifiers",
  "jailbreaking"
 ],
 "review": "assistant-drafted",
 "addedOn": "2026-09-25"
}