{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XIIL7R4UBZAEPCYLKLG4BEOILE","short_pith_number":"pith:XIIL7R4U","schema_version":"1.0","canonical_sha256":"ba10bfc7940e40478b0b52cdc091c859201465fb3f502a086a44e8da8584a202","source":{"kind":"arxiv","id":"2410.13708","version":2},"attestation_state":"computed","paper":{"title":"On the Role of Attention Heads in Large Language Model Safety","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Fei Huang, Haiyang Yu, Junfeng Fang, Kun Wang, Rongwu Xu, Xinghua Zhang, Yang Liu, Yongbin Li, Zhenhong Zhou","submitted_at":"2024-10-17T16:08:06Z","abstract_excerpt":"Large language models (LLMs) achieve state-of-the-art performance on multiple language tasks, yet their safety guardrails can be circumvented, leading to harmful generations. In light of this, recent research on safety mechanisms has emerged, revealing that when safety representations or component are suppressed, the safety capability of LLMs are compromised. However, existing research tends to overlook the safety impact of multi-head attention mechanisms, despite their crucial role in various model functionalities. Hence, in this paper, we aim to explore the connection between standard attent"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.13708","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-17T16:08:06Z","cross_cats_sorted":["cs.AI","cs.CR","cs.LG"],"title_canon_sha256":"194dc5a0d8c44030b738e852c1cfa25412ba1c88806e9375c62594e754e0c7be","abstract_canon_sha256":"4268a5157963e9812b49d7e1341762ab5786a81284c475925b5fa7357cffa2f7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:18:43.611013Z","signature_b64":"CDKrExyXtNlXXwmrE0RtvnjZHg5i7jeRC4TE+n5+9y8tqHYHQR1rYfCRAFS78pd6cAH1lcW/Tlgy5s8Ae01HDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ba10bfc7940e40478b0b52cdc091c859201465fb3f502a086a44e8da8584a202","last_reissued_at":"2026-07-05T10:18:43.610466Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:18:43.610466Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On the Role of Attention Heads in Large Language Model Safety","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Fei Huang, Haiyang Yu, Junfeng Fang, Kun Wang, Rongwu Xu, Xinghua Zhang, Yang Liu, Yongbin Li, Zhenhong Zhou","submitted_at":"2024-10-17T16:08:06Z","abstract_excerpt":"Large language models (LLMs) achieve state-of-the-art performance on multiple language tasks, yet their safety guardrails can be circumvented, leading to harmful generations. In light of this, recent research on safety mechanisms has emerged, revealing that when safety representations or component are suppressed, the safety capability of LLMs are compromised. However, existing research tends to overlook the safety impact of multi-head attention mechanisms, despite their crucial role in various model functionalities. Hence, in this paper, we aim to explore the connection between standard attent"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.13708","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.13708/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.13708","created_at":"2026-07-05T10:18:43.610528+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.13708v2","created_at":"2026-07-05T10:18:43.610528+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.13708","created_at":"2026-07-05T10:18:43.610528+00:00"},{"alias_kind":"pith_short_12","alias_value":"XIIL7R4UBZAE","created_at":"2026-07-05T10:18:43.610528+00:00"},{"alias_kind":"pith_short_16","alias_value":"XIIL7R4UBZAEPCYL","created_at":"2026-07-05T10:18:43.610528+00:00"},{"alias_kind":"pith_short_8","alias_value":"XIIL7R4U","created_at":"2026-07-05T10:18:43.610528+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.28153","citing_title":"Robust Harmful Features Under Jailbreak Attacks: Mechanistic Evidence from Attention Head Specialization in Large Language Models","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27997","citing_title":"Where Does Toxicity Live? Mechanistic Localization and Targeted Suppression in Language Models","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2506.02546","citing_title":"To trust or not to trust: Attention-based Trust Management for LLM Multi-Agent Systems","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2507.20906","citing_title":"Soft Head Selection for Injecting ICL-Derived Task Embeddings","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27401","citing_title":"Perturbation Probing: A Two-Pass-per-Prompt Diagnostic for FFN Behavioral Circuits in Aligned LLMs","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05668","citing_title":"Large Vision-Language Models Get Lost in Attention","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11663","citing_title":"Why Do Large Language Models Generate Harmful Content?","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11309","citing_title":"The Salami Slicing Threat: Exploiting Cumulative Risks in LLM Systems","ref_index":41,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XIIL7R4UBZAEPCYLKLG4BEOILE","json":"https://pith.science/pith/XIIL7R4UBZAEPCYLKLG4BEOILE.json","graph_json":"https://pith.science/api/pith-number/XIIL7R4UBZAEPCYLKLG4BEOILE/graph.json","events_json":"https://pith.science/api/pith-number/XIIL7R4UBZAEPCYLKLG4BEOILE/events.json","paper":"https://pith.science/paper/XIIL7R4U"},"agent_actions":{"view_html":"https://pith.science/pith/XIIL7R4UBZAEPCYLKLG4BEOILE","download_json":"https://pith.science/pith/XIIL7R4UBZAEPCYLKLG4BEOILE.json","view_paper":"https://pith.science/paper/XIIL7R4U","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.13708&json=true","fetch_graph":"https://pith.science/api/pith-number/XIIL7R4UBZAEPCYLKLG4BEOILE/graph.json","fetch_events":"https://pith.science/api/pith-number/XIIL7R4UBZAEPCYLKLG4BEOILE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XIIL7R4UBZAEPCYLKLG4BEOILE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XIIL7R4UBZAEPCYLKLG4BEOILE/action/storage_attestation","attest_author":"https://pith.science/pith/XIIL7R4UBZAEPCYLKLG4BEOILE/action/author_attestation","sign_citation":"https://pith.science/pith/XIIL7R4UBZAEPCYLKLG4BEOILE/action/citation_signature","submit_replication":"https://pith.science/pith/XIIL7R4UBZAEPCYLKLG4BEOILE/action/replication_record"}},"created_at":"2026-07-05T10:18:43.610528+00:00","updated_at":"2026-07-05T10:18:43.610528+00:00"}