{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:GMFJBMXXD32TYLTC6CQZTDH3AA","short_pith_number":"pith:GMFJBMXX","schema_version":"1.0","canonical_sha256":"330a90b2f71ef53c2e62f0a1998cfb00167774894c7ccb2dae23de76bd54fbd2","source":{"kind":"arxiv","id":"2406.05946","version":1},"attestation_state":"computed","paper":{"title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CR","authors_text":"Ahmad Beirami, Ashwinee Panda, Kaifeng Lyu, Peter Henderson, Prateek Mittal, Subhrajit Roy, Xiangyu Qi, Xiao Ma","submitted_at":"2024-06-10T00:35:23Z","abstract_excerpt":"The safety alignment of current Large Language Models (LLMs) is vulnerable. Relatively simple attacks, or even benign fine-tuning, can jailbreak aligned models. We argue that many of these vulnerabilities are related to a shared underlying issue: safety alignment can take shortcuts, wherein the alignment adapts a model's generative distribution primarily over only its very first few output tokens. We refer to this issue as shallow safety alignment. In this paper, we present case studies to explain why shallow safety alignment can exist and provide evidence that current aligned LLMs are subject"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.05946","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CR","submitted_at":"2024-06-10T00:35:23Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"9793c4fc1c6ef8cded44a19d5e978a2e6999f0ac5e4c6bd494a60f4d255772b8","abstract_canon_sha256":"d3c6ccfeb079a90f21d8eca1b6a852a83d177555f873e0ff32a2939a458a606b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:29:35.634382Z","signature_b64":"iIBPi+5OaHDK324l1jdrxjdFTfzr7xWbk3LC5leCHDyaVQPJRpsQXedU6Ba8ypsASrnajCt7TE9UaVaHaDmwAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"330a90b2f71ef53c2e62f0a1998cfb00167774894c7ccb2dae23de76bd54fbd2","last_reissued_at":"2026-07-05T08:29:35.633937Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:29:35.633937Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CR","authors_text":"Ahmad Beirami, Ashwinee Panda, Kaifeng Lyu, Peter Henderson, Prateek Mittal, Subhrajit Roy, Xiangyu Qi, Xiao Ma","submitted_at":"2024-06-10T00:35:23Z","abstract_excerpt":"The safety alignment of current Large Language Models (LLMs) is vulnerable. Relatively simple attacks, or even benign fine-tuning, can jailbreak aligned models. We argue that many of these vulnerabilities are related to a shared underlying issue: safety alignment can take shortcuts, wherein the alignment adapts a model's generative distribution primarily over only its very first few output tokens. We refer to this issue as shallow safety alignment. In this paper, we present case studies to explain why shallow safety alignment can exist and provide evidence that current aligned LLMs are subject"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.05946","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.05946/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.05946","created_at":"2026-07-05T08:29:35.633992+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.05946v1","created_at":"2026-07-05T08:29:35.633992+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.05946","created_at":"2026-07-05T08:29:35.633992+00:00"},{"alias_kind":"pith_short_12","alias_value":"GMFJBMXXD32T","created_at":"2026-07-05T08:29:35.633992+00:00"},{"alias_kind":"pith_short_16","alias_value":"GMFJBMXXD32TYLTC","created_at":"2026-07-05T08:29:35.633992+00:00"},{"alias_kind":"pith_short_8","alias_value":"GMFJBMXX","created_at":"2026-07-05T08:29:35.633992+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":31,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26515","citing_title":"Forget, Anticipate and Adapt: Test Time Training for Long Videos","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01239","citing_title":"Breaking Safety at the Token Boundary: How BPE Tokenization Creates Exploitable Gaps in LLM Alignment","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18673","citing_title":"Understanding and Mitigating Prompt Leaking Attacks in Real-World LLM-Based Applications","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18284","citing_title":"Breaking the Solver Bottleneck: Training Task Generators at the Learnable Frontier","ref_index":105,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08044","citing_title":"When Behavioral Safety Evaluation Fails: A Representation-Level Perspective","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04027","citing_title":"MaskForge: Structure-Aware Adaptive Attacks for Jailbreaking Diffusion Large Language Models","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01060","citing_title":"MENTIS: What Belief Changes Under Alignment? Measuring Multi-Scale Latent Torsion in Language Models","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30263","citing_title":"Defending Against Harmful Supervision Hidden in Benign Samples","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26515","citing_title":"Forget, Anticipate and Adapt: Test Time Training for Long Videos","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25603","citing_title":"Detecting Unfaithful Chain-of-Thought via Circuit-Guided Internal-External Discrepancy","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00651","citing_title":"MESA: Improving MoE Safety Alignment via Decentralized Expertise","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2409.18169","citing_title":"Harmful Fine-tuning Attacks and Defenses for Large Language Models: A Survey","ref_index":122,"is_internal_anchor":false},{"citing_arxiv_id":"2503.02574","citing_title":"LLM-Safety Evaluations Lack Robustness","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2602.07340","citing_title":"Revisiting Robustness for LLM Safety Alignment via Selective Geometry Control","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20342","citing_title":"ParaVT: Taming the Tool Prior Paradox for Parallel Tool Use in Agentic Video Reinforcement Learning","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20342","citing_title":"ParaVT: Taming the Tool Prior Paradox for Parallel Tool Use in Agentic Video Reinforcement Learning","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20654","citing_title":"REFLECTOR: Internalizing Step-wise Reflection against Indirect Jailbreak","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17413","citing_title":"Ablating Safety: Mechanisms for Removing Alignment in Language Models for Security Applications","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15239","citing_title":"Reducing the Safety Tax in LLM Safety Alignment with On-Policy Self-Distillation","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2506.05171","citing_title":"Towards provable probabilistic safety for scalable embodied AI systems","ref_index":138,"is_internal_anchor":false},{"citing_arxiv_id":"2508.04204","citing_title":"ReasoningGuard: Safeguarding Large Reasoning Models with Inference-time Safety Aha Moments","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2512.10998","citing_title":"SCOUT: A Defense Against Data Poisoning Attacks in Fine-Tuned Language Models","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2602.08813","citing_title":"Robust Policy Optimization to Prevent Catastrophic Forgetting","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12813","citing_title":"REALISTA: Realistic Latent Adversarial Attacks that Elicit LLM Hallucinations","ref_index":191,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08930","citing_title":"Internalizing Safety Understanding in Large Reasoning Models via Verification","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GMFJBMXXD32TYLTC6CQZTDH3AA","json":"https://pith.science/pith/GMFJBMXXD32TYLTC6CQZTDH3AA.json","graph_json":"https://pith.science/api/pith-number/GMFJBMXXD32TYLTC6CQZTDH3AA/graph.json","events_json":"https://pith.science/api/pith-number/GMFJBMXXD32TYLTC6CQZTDH3AA/events.json","paper":"https://pith.science/paper/GMFJBMXX"},"agent_actions":{"view_html":"https://pith.science/pith/GMFJBMXXD32TYLTC6CQZTDH3AA","download_json":"https://pith.science/pith/GMFJBMXXD32TYLTC6CQZTDH3AA.json","view_paper":"https://pith.science/paper/GMFJBMXX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.05946&json=true","fetch_graph":"https://pith.science/api/pith-number/GMFJBMXXD32TYLTC6CQZTDH3AA/graph.json","fetch_events":"https://pith.science/api/pith-number/GMFJBMXXD32TYLTC6CQZTDH3AA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GMFJBMXXD32TYLTC6CQZTDH3AA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GMFJBMXXD32TYLTC6CQZTDH3AA/action/storage_attestation","attest_author":"https://pith.science/pith/GMFJBMXXD32TYLTC6CQZTDH3AA/action/author_attestation","sign_citation":"https://pith.science/pith/GMFJBMXXD32TYLTC6CQZTDH3AA/action/citation_signature","submit_replication":"https://pith.science/pith/GMFJBMXXD32TYLTC6CQZTDH3AA/action/replication_record"}},"created_at":"2026-07-05T08:29:35.633992+00:00","updated_at":"2026-07-05T08:29:35.633992+00:00"}