{
  "node_id": "anthropic-responsible-scaling-policy-v2-1-2025",
  "title": "Anthropic Responsible Scaling Policy (Version 2.1, Effective 31 March 2025) - AI Safety Level Standards (ASL-2 Current Baseline; ASL-3 Required for Capability Thresholds in CBRN and Autonomous AI R&D); Capability Thresholds, Required Safeguards, and Governance Framework",
  "domain": "AI Governance & Law",
  "version": "1.0.0",
  "last_updated": "2025-03-31",
  "bluf": "Anthropic Responsible Scaling Policy version 2.1, effective 31 March 2025, is Anthropic PBC's public commitment not to train or deploy models capable of causing catastrophic harm unless safety and security measures keep risks below acceptable levels. The policy was first released in September 2023 (the original RSP) and updated in March 2025 to reflect lessons from the previous year. It is designed to be proportional (safeguards scale to risk), iterative (regularly measured and adjusted), and exportable (a prototype for other companies and a model for regulators). The core construct is AI Safety Level Standards (ASL Standards) - technical and operational measures for safely training and deploying frontier AI models - split into Deployment Standards and Security Standards. As of v2.1, all Anthropic models must meet the ASL-2 Deployment and Security Standards. The progression to ASL-3 (and beyond) is governed by Capability Thresholds and Required Safeguards: a Capability Threshold tells when protections must be upgraded, and the corresponding Required Safeguards specify what standard then applies. Version 2.1 provides specifications for Capability Thresholds in two domains: Chemical, Biological, Radiological, and Nuclear (CBRN) weapons, and Autonomous AI Research and Development (AI R&D), with the corresponding Required Safeguards identified. The capability assessment process is staged: a preliminary assessment first determines whether comprehensive evaluation is needed; comprehensive testing then evaluates whether the model is sufficiently below relevant Capability Thresholds absent surprising post-training enhancements; if Anthropic cannot make the required showing, it acts as though the model has surpassed the Threshold and upgrades to ASL-3 Required Safeguards while running follow-up assessment to confirm ASL-4 is not needed. The ASL-3 Deployment Standard requires robustness to persistent misuse attempts; the ASL-3 Security Standard requires being highly protected against non-state attackers attempting to steal model weights. Governance commitments include maintaining the position of Responsible Scaling Officer, an anonymous reporting channel for staff to notify the RSO of potential noncompliance, internal safety procedures for incident scenarios, and public release (with sensitive information removed) of key evaluation and deployment materials, soliciting input from external experts.",
  "paywall": {
    "status": "LOCKED",
    "unlock_cost_usd": "0.01",
    "skyfire_id": "41779894-ece2-4163-9761-b3b1b76e19b0"
  },
  "crosswalks": {
    "_available_keys": [
      "google-deepmind-frontier-safety-framework-v2-2025",
      "openai-preparedness-framework",
      "uk-aisi-ai-safety-evaluation-framework-2024",
      "us-aisi-ai-safety-institute-2024",
      "eu-ai-act-2024",
      "us-nist-sp-800-218a-secure-ai-development",
      "owasp-llm-top-10-2025"
    ],
    "_note": "Full crosswalk values included in vault response"
  },
  "dependencies": [
    "us-nist-sp-800-218a-secure-ai-development"
  ],
  "primary_citations_count": 13
}