{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:GL3GA2CTXZH4PZWJC4NIPXVB4S","short_pith_number":"pith:GL3GA2CT","schema_version":"1.0","canonical_sha256":"32f6606853be4fc7e6c9171a87dea1e4aa4ec753a25e5796c364c9d6a2b9db20","source":{"kind":"arxiv","id":"2412.16339","version":2},"attestation_state":"computed","paper":{"title":"Deliberative Alignment: Reasoning Enables Safer Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CY","cs.LG"],"primary_cat":"cs.CL","authors_text":"Alec Helyar, Alex Beutel, Amelia Glaese, Andrea Vallone, Boaz Barak, Eric Wallace, Hongyu Ren, Hyung Won Chung, Jason Wei, Johannes Heidecke, Manas Joglekar, Melody Y. Guan, Rachel Dias, Saachi Jain, Sam Toyer","submitted_at":"2024-12-20T21:00:11Z","abstract_excerpt":"As large-scale language models increasingly impact safety-critical domains, ensuring their reliable adherence to well-defined principles remains a fundamental challenge. We introduce Deliberative Alignment, a new paradigm that directly teaches the model safety specifications and trains it to explicitly recall and accurately reason over the specifications before answering. We used this approach to align OpenAI's o-series models, and achieved highly precise adherence to OpenAI's safety policies, without requiring human-written chain-of-thoughts or answers. Deliberative Alignment pushes the Paret"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.16339","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-12-20T21:00:11Z","cross_cats_sorted":["cs.AI","cs.CY","cs.LG"],"title_canon_sha256":"169e485ad16ae2390ab9ddae7818c25014c6ef2fcc51fd78fa7fa508358d847a","abstract_canon_sha256":"c52d4158d362cd27a0e7bf6365d16f54892e5be54acefff3b0b5ea74a056c5c3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:58:44.313115Z","signature_b64":"bTIMyYqnOrLNr0fsOs54HInaO4YBLE3BMcQD2bPzo1HCg9Fad2F2egRDlQZJDiCmCVNnjYlsrAp3hXg4GHllCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"32f6606853be4fc7e6c9171a87dea1e4aa4ec753a25e5796c364c9d6a2b9db20","last_reissued_at":"2026-07-05T09:58:44.312571Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:58:44.312571Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Deliberative Alignment: Reasoning Enables Safer Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CY","cs.LG"],"primary_cat":"cs.CL","authors_text":"Alec Helyar, Alex Beutel, Amelia Glaese, Andrea Vallone, Boaz Barak, Eric Wallace, Hongyu Ren, Hyung Won Chung, Jason Wei, Johannes Heidecke, Manas Joglekar, Melody Y. Guan, Rachel Dias, Saachi Jain, Sam Toyer","submitted_at":"2024-12-20T21:00:11Z","abstract_excerpt":"As large-scale language models increasingly impact safety-critical domains, ensuring their reliable adherence to well-defined principles remains a fundamental challenge. We introduce Deliberative Alignment, a new paradigm that directly teaches the model safety specifications and trains it to explicitly recall and accurately reason over the specifications before answering. We used this approach to align OpenAI's o-series models, and achieved highly precise adherence to OpenAI's safety policies, without requiring human-written chain-of-thoughts or answers. Deliberative Alignment pushes the Paret"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.16339","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.16339/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.16339","created_at":"2026-07-05T09:58:44.312629+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.16339v2","created_at":"2026-07-05T09:58:44.312629+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.16339","created_at":"2026-07-05T09:58:44.312629+00:00"},{"alias_kind":"pith_short_12","alias_value":"GL3GA2CTXZH4","created_at":"2026-07-05T09:58:44.312629+00:00"},{"alias_kind":"pith_short_16","alias_value":"GL3GA2CTXZH4PZWJ","created_at":"2026-07-05T09:58:44.312629+00:00"},{"alias_kind":"pith_short_8","alias_value":"GL3GA2CT","created_at":"2026-07-05T09:58:44.312629+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":32,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25442","citing_title":"PolicyAlign: Direct Policy-Based Safety Alignment for Large Language Models","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.25013","citing_title":"Do Thinking Tokens Help with Safety?","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20470","citing_title":"Analyzing Defensive Misdirection Against Model-Guided Automated Attacks on Agentic AI Systems","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01277","citing_title":"Cognitive Firewall: A Proactive, Zero-Trust, Multi-Gate Framework for LLM Safety","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.15531","citing_title":"Greedy Coordinate Diffusion: Effective and Semantically Coherent Adversarial Attacks via Diffusion Guidance","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09711","citing_title":"Proxy Reward Internalization and Mechanistic Exploitation: A Learned Precursor to Reward Hacking and Its Generalization","ref_index":244,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08682","citing_title":"Activation Steering Induces Emergent Misalignment: A More Comprehensive Evaluation","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00572","citing_title":"HARC: Coupling Harmfulness and Refusal Directions for Robust Safety Alignment","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24834","citing_title":"Reflect-Guard: Enhancing LLM Safeguards against Adversarial Prompts via Logical Self-Reflection","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24229","citing_title":"How Well Do Models Follow Their Constitutions?","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20470","citing_title":"Analyzing Defensive Misdirection Against Model-Guided Automated Attacks on Agentic AI Systems","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28739","citing_title":"Agent Safety Is Action Alignment","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28467","citing_title":"Mitigating Adaptive Attacks against Reasoning Models with Activation Consistency Training","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06840","citing_title":"Characterize Then Distill: Mechanistic Reasoning in Large Output Spaces","ref_index":100,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23565","citing_title":"Understanding Goal Generalisation in Sequential Reinforcement Learning","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2503.02574","citing_title":"LLM-Safety Evaluations Lack Robustness","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2603.17305","citing_title":"Contrastive Reasoning Alignment: Reinforcement Learning from Hidden Representations","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2503.11926","citing_title":"Monitoring Reasoning Models for Misbehavior and the Risks of Promoting Obfuscation","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2509.05367","citing_title":"Between a Rock and a Hard Place: The Tension Between Ethical Reasoning and Safety Alignment in LLMs","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2602.07892","citing_title":"Safety Alignment as Continual Learning: Mitigating the Alignment Tax via Orthogonal Gradient Projection","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2502.13189","citing_title":"MoBA: Mixture of Block Attention for Long-Context LLMs","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05630","citing_title":"One Turn Too Late: Response-Aware Defense Against Hidden Malicious Intent in Multi-Turn Dialogue","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08930","citing_title":"Internalizing Safety Understanding in Large Reasoning Models via Verification","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03900","citing_title":"Contextual Multi-Objective Optimization: Rethinking Objectives in Frontier AI Systems","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05630","citing_title":"One Turn Too Late: Response-Aware Defense Against Hidden Malicious Intent in Multi-Turn Dialogue","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GL3GA2CTXZH4PZWJC4NIPXVB4S","json":"https://pith.science/pith/GL3GA2CTXZH4PZWJC4NIPXVB4S.json","graph_json":"https://pith.science/api/pith-number/GL3GA2CTXZH4PZWJC4NIPXVB4S/graph.json","events_json":"https://pith.science/api/pith-number/GL3GA2CTXZH4PZWJC4NIPXVB4S/events.json","paper":"https://pith.science/paper/GL3GA2CT"},"agent_actions":{"view_html":"https://pith.science/pith/GL3GA2CTXZH4PZWJC4NIPXVB4S","download_json":"https://pith.science/pith/GL3GA2CTXZH4PZWJC4NIPXVB4S.json","view_paper":"https://pith.science/paper/GL3GA2CT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.16339&json=true","fetch_graph":"https://pith.science/api/pith-number/GL3GA2CTXZH4PZWJC4NIPXVB4S/graph.json","fetch_events":"https://pith.science/api/pith-number/GL3GA2CTXZH4PZWJC4NIPXVB4S/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GL3GA2CTXZH4PZWJC4NIPXVB4S/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GL3GA2CTXZH4PZWJC4NIPXVB4S/action/storage_attestation","attest_author":"https://pith.science/pith/GL3GA2CTXZH4PZWJC4NIPXVB4S/action/author_attestation","sign_citation":"https://pith.science/pith/GL3GA2CTXZH4PZWJC4NIPXVB4S/action/citation_signature","submit_replication":"https://pith.science/pith/GL3GA2CTXZH4PZWJC4NIPXVB4S/action/replication_record"}},"created_at":"2026-07-05T09:58:44.312629+00:00","updated_at":"2026-07-05T09:58:44.312629+00:00"}