{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:WR42A6MNMIEANQZLMTRK3JZIER","short_pith_number":"pith:WR42A6MN","schema_version":"1.0","canonical_sha256":"b479a0798d620806c32b64e2ada728245f93bf083a061720d71b265bcc398b7e","source":{"kind":"arxiv","id":"2409.01586","version":4},"attestation_state":"computed","paper":{"title":"Booster: Tackling Harmful Fine-tuning for Large Language Models via Attenuating Harmful Perturbation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Fatih Ilhan, Ling Liu, Selim Furkan Tekin, Sihao Hu, Tiansheng Huang","submitted_at":"2024-09-03T03:59:22Z","abstract_excerpt":"Harmful fine-tuning attack poses serious safety concerns for large language models' fine-tuning-as-a-service. While existing defenses have been proposed to mitigate the issue, their performances are still far away from satisfactory, and the root cause of the problem has not been fully recovered. To this end, we in this paper show that harmful perturbation over the model weights could be a probable cause of alignment-broken. In order to attenuate the negative impact of harmful perturbation, we propose an alignment-stage solution, dubbed Booster. Technically, along with the original alignment lo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.01586","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-09-03T03:59:22Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"a383b97a494deb9199879819b323d3824c71cefa662fdf0b1a70b94cb3dbf073","abstract_canon_sha256":"599300368443dddddbc6402b408d5ab8c60ca2161753c4b161fe8a072360ef8b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:32:29.973400Z","signature_b64":"7Bru06bJmgL98lvD6Ju1W3V3+XB+vecCektJoRcCeeyNm8I74h3I+8fMZRzkSuKRa1PT6qgXxEmeSpSfHwqcDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b479a0798d620806c32b64e2ada728245f93bf083a061720d71b265bcc398b7e","last_reissued_at":"2026-07-05T10:32:29.972632Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:32:29.972632Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Booster: Tackling Harmful Fine-tuning for Large Language Models via Attenuating Harmful Perturbation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Fatih Ilhan, Ling Liu, Selim Furkan Tekin, Sihao Hu, Tiansheng Huang","submitted_at":"2024-09-03T03:59:22Z","abstract_excerpt":"Harmful fine-tuning attack poses serious safety concerns for large language models' fine-tuning-as-a-service. While existing defenses have been proposed to mitigate the issue, their performances are still far away from satisfactory, and the root cause of the problem has not been fully recovered. To this end, we in this paper show that harmful perturbation over the model weights could be a probable cause of alignment-broken. In order to attenuate the negative impact of harmful perturbation, we propose an alignment-stage solution, dubbed Booster. Technically, along with the original alignment lo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.01586","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.01586/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.01586","created_at":"2026-07-05T10:32:29.972727+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.01586v4","created_at":"2026-07-05T10:32:29.972727+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.01586","created_at":"2026-07-05T10:32:29.972727+00:00"},{"alias_kind":"pith_short_12","alias_value":"WR42A6MNMIEA","created_at":"2026-07-05T10:32:29.972727+00:00"},{"alias_kind":"pith_short_16","alias_value":"WR42A6MNMIEANQZL","created_at":"2026-07-05T10:32:29.972727+00:00"},{"alias_kind":"pith_short_8","alias_value":"WR42A6MN","created_at":"2026-07-05T10:32:29.972727+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.14605","citing_title":"One Step to the Side: Why Defenses Against Malicious Finetuning Fail Under Adaptive Adversaries","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28030","citing_title":"SPARD: Defending Harmful Fine-Tuning Attack via Safety Projection with Relevance-Diversity Data Selection","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2409.18169","citing_title":"Harmful Fine-tuning Attacks and Defenses for Large Language Models: A Survey","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2502.05206","citing_title":"Safety at Scale: A Comprehensive Survey of Large Model and Agent Safety","ref_index":121,"is_internal_anchor":false},{"citing_arxiv_id":"2505.16737","citing_title":"Secure LLM Fine-Tuning via Safety-Aware Probing","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2508.20697","citing_title":"Token Buncher: Shielding LLMs from Harmful Reinforcement Learning Fine-Tuning","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09927","citing_title":"Information Extraction of Nested Complex Structure of Quantum Cascade Lasers via Large Language Models","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19016","citing_title":"AlignCultura: Towards Culturally Aligned Large Language Models?","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12384","citing_title":"Preventing Safety Drift in Large Language Models via Coupled Weight and Activation Constraints","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WR42A6MNMIEANQZLMTRK3JZIER","json":"https://pith.science/pith/WR42A6MNMIEANQZLMTRK3JZIER.json","graph_json":"https://pith.science/api/pith-number/WR42A6MNMIEANQZLMTRK3JZIER/graph.json","events_json":"https://pith.science/api/pith-number/WR42A6MNMIEANQZLMTRK3JZIER/events.json","paper":"https://pith.science/paper/WR42A6MN"},"agent_actions":{"view_html":"https://pith.science/pith/WR42A6MNMIEANQZLMTRK3JZIER","download_json":"https://pith.science/pith/WR42A6MNMIEANQZLMTRK3JZIER.json","view_paper":"https://pith.science/paper/WR42A6MN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.01586&json=true","fetch_graph":"https://pith.science/api/pith-number/WR42A6MNMIEANQZLMTRK3JZIER/graph.json","fetch_events":"https://pith.science/api/pith-number/WR42A6MNMIEANQZLMTRK3JZIER/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WR42A6MNMIEANQZLMTRK3JZIER/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WR42A6MNMIEANQZLMTRK3JZIER/action/storage_attestation","attest_author":"https://pith.science/pith/WR42A6MNMIEANQZLMTRK3JZIER/action/author_attestation","sign_citation":"https://pith.science/pith/WR42A6MNMIEANQZLMTRK3JZIER/action/citation_signature","submit_replication":"https://pith.science/pith/WR42A6MNMIEANQZLMTRK3JZIER/action/replication_record"}},"created_at":"2026-07-05T10:32:29.972727+00:00","updated_at":"2026-07-05T10:32:29.972727+00:00"}