{
  "node_id": "google-deepmind-frontier-safety-framework-v2-2025",
  "title": "Google DeepMind Frontier Safety Framework - Second Iteration (4 February 2025) - Critical Capability Levels (CCLs), Security Level Recommendations, Updated Deployment Mitigation Procedure, and Industry-Leading Approach to Deceptive Alignment Risk",
  "domain": "AI Governance & Law",
  "version": "1.0.0",
  "last_updated": "2025-02-04",
  "bluf": "The Google DeepMind Frontier Safety Framework (FSF), second iteration published on 4 February 2025, is Google DeepMind's authoritative public framework for staying ahead of possible severe risks from powerful frontier AI models. The first iteration was introduced in May 2024; the second iteration (this version) was published in February 2025; subsequent strengthening was published in September 2025 ('Strengthening our Frontier Safety Framework'). The Framework has been implemented in DeepMind's safety and governance processes for evaluating frontier models such as Gemini 2.0. The core construct is Critical Capability Levels (CCLs) - the minimum level of capabilities a model must have to play a role in causing severe harm, identified by researching the paths through which a model could cause severe harm in high-risk domains and determining the threshold capabilities for each. Three key updates distinguish the second iteration: (1) Security Level recommendations for each of DeepMind's CCLs, identifying where the strongest efforts to curb model-weight exfiltration risk are needed - with particularly high security levels recommended for CCLs in the domain of machine learning research and development (R&D) given the risk of uncontrolled proliferation accelerating AI development; (2) a more consistent procedure for applying deployment mitigations - preparing a set of mitigations through iterative safeguards development, building an assessable safety case showing severe risks have been minimised to acceptable levels, with the appropriate corporate governance body reviewing the safety case and general availability deployment occurring only if approved, with continued post-deployment review and update; (3) an industry-leading approach to deceptive alignment risk - addressing the risk of an autonomous system deliberately undermining human control, initially by detecting baseline instrumental reasoning ability through automated monitoring and committing to further research as models reach stronger instrumental-reasoning capabilities. The Framework commits to sharing information with appropriate government authorities where a model is assessed to have reached a CCL posing unmitigated and material risk to public safety. Authors include Lewis Ho, Celine Smith, Claudia van der Salm, Joslyn Barnhart, Rohin Shah; leadership Allan Dafoe, Anca Dragan, Andy Song, Demis Hassabis, Four Flynn, Jennifer Beroshi, Helen King, Nicklas Lundblad, and Tom Lue. The Framework is anchored on Google's broader AI Principles and intersects with the Seoul Frontier AI Safety Commitments.",
  "paywall": {
    "status": "LOCKED",
    "unlock_cost_usd": "0.01",
    "skyfire_id": "41779894-ece2-4163-9761-b3b1b76e19b0"
  },
  "crosswalks": {
    "_available_keys": [
      "anthropic-responsible-scaling-policy-v2-1-2025",
      "openai-preparedness-framework",
      "uk-aisi-ai-safety-evaluation-framework-2024",
      "us-aisi-ai-safety-institute-2024",
      "seoul-frontier-ai-safety-commitments-2024",
      "eu-ai-act-2024",
      "us-nist-sp-800-218a-secure-ai-development"
    ],
    "_note": "Full crosswalk values included in vault response"
  },
  "dependencies": [
    "anthropic-responsible-scaling-policy-v2-1-2025",
    "us-nist-sp-800-218a-secure-ai-development"
  ],
  "primary_citations_count": 12
}