{"work":{"id":"0adadb60-1c1e-4c3f-ba68-7e500b454d96","openalex_id":"https://openalex.org/W4391709278","doi":"10.48550/arxiv.2402.05162","arxiv_id":"2402.05162","raw_key":null,"title":"Assessing the Brittleness of Safety Alignment via Pruning and Low-Rank Modifications","authors":null,"authors_text":"Assessing the brittleness of safety alignment via pruning and low-rank modifications , author=","year":2024,"venue":"cs.LG","abstract":"Large language models (LLMs) show inherent brittleness in their safety mechanisms, as evidenced by their susceptibility to jailbreaking and even non-malicious fine-tuning. This study explores this brittleness of safety alignment by leveraging pruning and low-rank modifications. We develop methods to identify critical regions that are vital for safety guardrails, and that are disentangled from utility-relevant regions at both the neuron and rank levels. Surprisingly, the isolated regions we find are sparse, comprising about $3\\%$ at the parameter level and $2.5\\%$ at the rank level. Removing these regions compromises safety without significantly impacting utility, corroborating the inherent brittleness of the model's safety mechanisms. Moreover, we show that LLMs remain vulnerable to low-cost fine-tuning attacks even when modifications to the safety-critical regions are restricted. These findings underscore the urgent need for more robust safety strategies in LLMs.","external_url":"https://arxiv.org/abs/2402.05162","cited_by_count":2,"metadata_source":"pith","metadata_fetched_at":"2026-08-05T02:28:24.338817+00:00","pith_arxiv_id":"2402.05162","created_at":"2026-05-11T09:21:01.736370+00:00","updated_at":"2026-08-05T02:28:24.338817+00:00","title_quality_ok":true,"display_title":"Assessing the brittleness of safety alignment via pruning and low-rank modifications","render_title":"Assessing the brittleness of safety alignment via pruning and low-rank modifications"},"hub":{"state":{"work_id":"0adadb60-1c1e-4c3f-ba68-7e500b454d96","tier":"hub","tier_reason":"10+ Pith inbound or 1,000+ external citations","pith_inbound_count":15,"external_cited_by_count":2,"distinct_field_count":6,"first_pith_cited_at":"2024-06-17T16:36:12+00:00","last_pith_cited_at":"2026-07-06T17:33:36+00:00","author_build_status":"not_needed","summary_status":"needed","contexts_status":"needed","graph_status":"needed","ask_index_status":"not_needed","reader_status":"not_needed","recognition_status":"not_needed","updated_at":"2026-08-22T15:59:49.817346+00:00","tier_text":"hub"},"tier":"hub","role_counts":[{"context_role":"background","n":1}],"polarity_counts":[{"context_polarity":"background","n":1}],"runs":{},"summary":{},"graph":{},"authors":[]}}