{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:K6HERLGFR7HOH545OQLUHLCO52","short_pith_number":"pith:K6HERLGF","schema_version":"1.0","canonical_sha256":"578e48acc58fcee3f79d741743ac4eee9016b053729a5b46cd380cfd20596aa2","source":{"kind":"arxiv","id":"2506.15606","version":3},"attestation_state":"computed","paper":{"title":"LoX: Low-Rank Extrapolation Robustifies LLM Safety Against Fine-tuning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Gabriel J. Perin, Junyuan Hong, Nina S. T. Hirata, Runjin Chen, Xuxi Chen, Zhangyang Wang","submitted_at":"2025-06-18T16:30:02Z","abstract_excerpt":"Large Language Models (LLMs) have become indispensable in real-world applications. However, their widespread adoption raises significant safety concerns, particularly in responding to socially harmful questions. Despite substantial efforts to improve model safety through alignment, aligned models can still have their safety protections undermined by subsequent fine-tuning - even when the additional training data appears benign. In this paper, we empirically demonstrate that this vulnerability stems from the sensitivity of safety-critical low-rank subspaces in LLM parameters to fine-tuning. Bui"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.15606","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-06-18T16:30:02Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"8d6ff878735c21e682158644c9870fefa80792bc5654fda537d65a3539bc8609","abstract_canon_sha256":"bfe7ec4848da87492e264eff305cae73599c5e288ca21f527e004b246c888989"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:43:32.908405Z","signature_b64":"VkaRTv7DO2tRXdrOKg51SMQ0/+y6ES4EE04KEXzxqhSsuABg87gA5SCzoXuly5VMHm5biCm1Xfs0bkuWnpaiAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"578e48acc58fcee3f79d741743ac4eee9016b053729a5b46cd380cfd20596aa2","last_reissued_at":"2026-07-05T11:43:32.907940Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:43:32.907940Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LoX: Low-Rank Extrapolation Robustifies LLM Safety Against Fine-tuning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Gabriel J. Perin, Junyuan Hong, Nina S. T. Hirata, Runjin Chen, Xuxi Chen, Zhangyang Wang","submitted_at":"2025-06-18T16:30:02Z","abstract_excerpt":"Large Language Models (LLMs) have become indispensable in real-world applications. However, their widespread adoption raises significant safety concerns, particularly in responding to socially harmful questions. Despite substantial efforts to improve model safety through alignment, aligned models can still have their safety protections undermined by subsequent fine-tuning - even when the additional training data appears benign. In this paper, we empirically demonstrate that this vulnerability stems from the sensitivity of safety-critical low-rank subspaces in LLM parameters to fine-tuning. Bui"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.15606","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.15606/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.15606","created_at":"2026-07-05T11:43:32.907994+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.15606v3","created_at":"2026-07-05T11:43:32.907994+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.15606","created_at":"2026-07-05T11:43:32.907994+00:00"},{"alias_kind":"pith_short_12","alias_value":"K6HERLGFR7HO","created_at":"2026-07-05T11:43:32.907994+00:00"},{"alias_kind":"pith_short_16","alias_value":"K6HERLGFR7HOH545","created_at":"2026-07-05T11:43:32.907994+00:00"},{"alias_kind":"pith_short_8","alias_value":"K6HERLGF","created_at":"2026-07-05T11:43:32.907994+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.14605","citing_title":"One Step to the Side: Why Defenses Against Malicious Finetuning Fail Under Adaptive Adversaries","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2602.07340","citing_title":"Revisiting Robustness for LLM Safety Alignment via Selective Geometry Control","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12384","citing_title":"Preventing Safety Drift in Large Language Models via Coupled Weight and Activation Constraints","ref_index":32,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/K6HERLGFR7HOH545OQLUHLCO52","json":"https://pith.science/pith/K6HERLGFR7HOH545OQLUHLCO52.json","graph_json":"https://pith.science/api/pith-number/K6HERLGFR7HOH545OQLUHLCO52/graph.json","events_json":"https://pith.science/api/pith-number/K6HERLGFR7HOH545OQLUHLCO52/events.json","paper":"https://pith.science/paper/K6HERLGF"},"agent_actions":{"view_html":"https://pith.science/pith/K6HERLGFR7HOH545OQLUHLCO52","download_json":"https://pith.science/pith/K6HERLGFR7HOH545OQLUHLCO52.json","view_paper":"https://pith.science/paper/K6HERLGF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.15606&json=true","fetch_graph":"https://pith.science/api/pith-number/K6HERLGFR7HOH545OQLUHLCO52/graph.json","fetch_events":"https://pith.science/api/pith-number/K6HERLGFR7HOH545OQLUHLCO52/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/K6HERLGFR7HOH545OQLUHLCO52/action/timestamp_anchor","attest_storage":"https://pith.science/pith/K6HERLGFR7HOH545OQLUHLCO52/action/storage_attestation","attest_author":"https://pith.science/pith/K6HERLGFR7HOH545OQLUHLCO52/action/author_attestation","sign_citation":"https://pith.science/pith/K6HERLGFR7HOH545OQLUHLCO52/action/citation_signature","submit_replication":"https://pith.science/pith/K6HERLGFR7HOH545OQLUHLCO52/action/replication_record"}},"created_at":"2026-07-05T11:43:32.907994+00:00","updated_at":"2026-07-05T11:43:32.907994+00:00"}