{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:YY35Y3FSTFQ5RWHQRIQODERNQ4","short_pith_number":"pith:YY35Y3FS","schema_version":"1.0","canonical_sha256":"c637dc6cb29961d8d8f08a20e1922d871dee6faa0d10b342efac70ec4245cb06","source":{"kind":"arxiv","id":"2501.01765","version":1},"attestation_state":"computed","paper":{"title":"SaLoRA: Safety-Alignment Preserved Low-Rank Adaptation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Michael Backes, Mingjie Li, Wai Man Si, Yang Zhang, Yisen Wang","submitted_at":"2025-01-03T11:34:28Z","abstract_excerpt":"As advancements in large language models (LLMs) continue and the demand for personalized models increases, parameter-efficient fine-tuning (PEFT) methods (e.g., LoRA) will become essential due to their efficiency in reducing computation costs. However, recent studies have raised alarming concerns that LoRA fine-tuning could potentially compromise the safety alignment in LLMs, posing significant risks for the model owner. In this paper, we first investigate the underlying mechanism by analyzing the changes in safety alignment related features before and after fine-tuning. Then, we propose a fix"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.01765","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-01-03T11:34:28Z","cross_cats_sorted":[],"title_canon_sha256":"b8ca6e1ae1931839716f2c576be5d9e78ff1969406d1e9e13fcd85017f497ab6","abstract_canon_sha256":"481789ca7aeca4f4f9bfda27db4afb99dadc9a100b09823e29ce0dc1c3b23a3a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:56:40.571907Z","signature_b64":"ExJyzxjHHiT1miQ/KJSCHVZhFFePtcHzvRIrwzXxGqzvowhDJLxey5gW1ZtdxR746KtOKPhwW0kHn7rOP/KPDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c637dc6cb29961d8d8f08a20e1922d871dee6faa0d10b342efac70ec4245cb06","last_reissued_at":"2026-07-05T09:56:40.571413Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:56:40.571413Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SaLoRA: Safety-Alignment Preserved Low-Rank Adaptation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Michael Backes, Mingjie Li, Wai Man Si, Yang Zhang, Yisen Wang","submitted_at":"2025-01-03T11:34:28Z","abstract_excerpt":"As advancements in large language models (LLMs) continue and the demand for personalized models increases, parameter-efficient fine-tuning (PEFT) methods (e.g., LoRA) will become essential due to their efficiency in reducing computation costs. However, recent studies have raised alarming concerns that LoRA fine-tuning could potentially compromise the safety alignment in LLMs, posing significant risks for the model owner. In this paper, we first investigate the underlying mechanism by analyzing the changes in safety alignment related features before and after fine-tuning. Then, we propose a fix"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.01765","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.01765/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.01765","created_at":"2026-07-05T09:56:40.571494+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.01765v1","created_at":"2026-07-05T09:56:40.571494+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.01765","created_at":"2026-07-05T09:56:40.571494+00:00"},{"alias_kind":"pith_short_12","alias_value":"YY35Y3FSTFQ5","created_at":"2026-07-05T09:56:40.571494+00:00"},{"alias_kind":"pith_short_16","alias_value":"YY35Y3FSTFQ5RWHQ","created_at":"2026-07-05T09:56:40.571494+00:00"},{"alias_kind":"pith_short_8","alias_value":"YY35Y3FS","created_at":"2026-07-05T09:56:40.571494+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.06519","citing_title":"SafeGene: Reusable Adapters for Transferable Safety Alignment","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00160","citing_title":"DataShield: Safety-degrading Data Filtering for LLM Benign Instruction Fine-Tuning","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28030","citing_title":"SPARD: Defending Harmful Fine-Tuning Attack via Safety Projection with Relevance-Diversity Data Selection","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2505.01307","citing_title":"Document Retrieval Augmented Fine-Tuning (DRAFT) for safety-critical software assessments","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04992","citing_title":"You Snooze, You Lose: Automatic Safety Alignment Restoration through Neural Weight Translation","ref_index":81,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21905","citing_title":"Low-Rank Adaptation Redux for Large Models","ref_index":108,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12384","citing_title":"Preventing Safety Drift in Large Language Models via Coupled Weight and Activation Constraints","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YY35Y3FSTFQ5RWHQRIQODERNQ4","json":"https://pith.science/pith/YY35Y3FSTFQ5RWHQRIQODERNQ4.json","graph_json":"https://pith.science/api/pith-number/YY35Y3FSTFQ5RWHQRIQODERNQ4/graph.json","events_json":"https://pith.science/api/pith-number/YY35Y3FSTFQ5RWHQRIQODERNQ4/events.json","paper":"https://pith.science/paper/YY35Y3FS"},"agent_actions":{"view_html":"https://pith.science/pith/YY35Y3FSTFQ5RWHQRIQODERNQ4","download_json":"https://pith.science/pith/YY35Y3FSTFQ5RWHQRIQODERNQ4.json","view_paper":"https://pith.science/paper/YY35Y3FS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.01765&json=true","fetch_graph":"https://pith.science/api/pith-number/YY35Y3FSTFQ5RWHQRIQODERNQ4/graph.json","fetch_events":"https://pith.science/api/pith-number/YY35Y3FSTFQ5RWHQRIQODERNQ4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YY35Y3FSTFQ5RWHQRIQODERNQ4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YY35Y3FSTFQ5RWHQRIQODERNQ4/action/storage_attestation","attest_author":"https://pith.science/pith/YY35Y3FSTFQ5RWHQRIQODERNQ4/action/author_attestation","sign_citation":"https://pith.science/pith/YY35Y3FSTFQ5RWHQRIQODERNQ4/action/citation_signature","submit_replication":"https://pith.science/pith/YY35Y3FSTFQ5RWHQRIQODERNQ4/action/replication_record"}},"created_at":"2026-07-05T09:56:40.571494+00:00","updated_at":"2026-07-05T09:56:40.571494+00:00"}