{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ASSWOMG3RWDG4VY4YVLT4H7VCV","short_pith_number":"pith:ASSWOMG3","schema_version":"1.0","canonical_sha256":"04a56730db8d866e571cc5573e1ff5155ef2321465a443eb5fa9dbc7c42d1449","source":{"kind":"arxiv","id":"2408.00761","version":4},"attestation_state":"computed","paper":{"title":"Tamper-Resistant Safeguards for Open-Weight LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Alice Gatti, Andy Zhou, Andy Zou, Bhrugu Bharathi, Bo Li, Dan Hendrycks, Dawn Song, Justin Wang, Long Phan, Mantas Mazeika, Maxwell Lin, Rishub Tamirisa, Ron Arel, Rowan Wang, Tarun Suresh","submitted_at":"2024-08-01T17:59:12Z","abstract_excerpt":"Rapid advances in the capabilities of large language models (LLMs) have raised widespread concerns regarding their potential for malicious use. Open-weight LLMs present unique challenges, as existing safeguards lack robustness to tampering attacks that modify model weights. For example, recent works have demonstrated that refusal and unlearning safeguards can be trivially removed with a few steps of fine-tuning. These vulnerabilities necessitate new approaches for enabling the safe release of open-weight LLMs. We develop a method, called TAR, for building tamper-resistant safeguards into open-"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.00761","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-08-01T17:59:12Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"70573cc4b6dce644ec0c38f18f643a2c9ff0ed9e6e1c1d1e692c9a4e1c629d1b","abstract_canon_sha256":"0fe46027c91b0e8578b6e6461d93a8b2f0631a94672f975a9e3fff27e27ec259"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:12:00.270430Z","signature_b64":"xE/NXLTtsVDVqH3hUDBW9qmGIonPF8lVU1v110pqaA8DgeFnIBwTJtIi1vChneY+1CvqbZXCmMS7YPX1h42ECg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"04a56730db8d866e571cc5573e1ff5155ef2321465a443eb5fa9dbc7c42d1449","last_reissued_at":"2026-07-05T10:12:00.269902Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:12:00.269902Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Tamper-Resistant Safeguards for Open-Weight LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Alice Gatti, Andy Zhou, Andy Zou, Bhrugu Bharathi, Bo Li, Dan Hendrycks, Dawn Song, Justin Wang, Long Phan, Mantas Mazeika, Maxwell Lin, Rishub Tamirisa, Ron Arel, Rowan Wang, Tarun Suresh","submitted_at":"2024-08-01T17:59:12Z","abstract_excerpt":"Rapid advances in the capabilities of large language models (LLMs) have raised widespread concerns regarding their potential for malicious use. Open-weight LLMs present unique challenges, as existing safeguards lack robustness to tampering attacks that modify model weights. For example, recent works have demonstrated that refusal and unlearning safeguards can be trivially removed with a few steps of fine-tuning. These vulnerabilities necessitate new approaches for enabling the safe release of open-weight LLMs. We develop a method, called TAR, for building tamper-resistant safeguards into open-"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.00761","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.00761/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.00761","created_at":"2026-07-05T10:12:00.269968+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.00761v4","created_at":"2026-07-05T10:12:00.269968+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.00761","created_at":"2026-07-05T10:12:00.269968+00:00"},{"alias_kind":"pith_short_12","alias_value":"ASSWOMG3RWDG","created_at":"2026-07-05T10:12:00.269968+00:00"},{"alias_kind":"pith_short_16","alias_value":"ASSWOMG3RWDG4VY4","created_at":"2026-07-05T10:12:00.269968+00:00"},{"alias_kind":"pith_short_8","alias_value":"ASSWOMG3","created_at":"2026-07-05T10:12:00.269968+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17168","citing_title":"RepSelect: Robust LLM Unlearning via Representation Selectivity","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14605","citing_title":"One Step to the Side: Why Defenses Against Malicious Finetuning Fail Under Adaptive Adversaries","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28962","citing_title":"FlipGuard: Defending Large Language Models Against Quantization-Conditioned Backdoor Attacks","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29239","citing_title":"Breaking the Rounding Trap: Securing LLMs against Quantization-Conditioned Backdoors","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2409.18169","citing_title":"Harmful Fine-tuning Attacks and Defenses for Large Language Models: A Survey","ref_index":144,"is_internal_anchor":false},{"citing_arxiv_id":"2505.16737","citing_title":"Secure LLM Fine-Tuning via Safety-Aware Probing","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2510.00761","citing_title":"Downgrade to Upgrade: Optimizer Simplification Enhances Robustness in LLM Unlearning","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2511.22681","citing_title":"CacheTrap: Unveiling a Stealthier Gray-Box Trojan against LLMs","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2602.08813","citing_title":"Robust Policy Optimization to Prevent Catastrophic Forgetting","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11685","citing_title":"Robust LLM Unlearning Against Relearning Attacks: The Minor Components in Representations Matter","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24902","citing_title":"Safety Drift After Fine-Tuning: Evidence from High-Stakes Domains","ref_index":56,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ASSWOMG3RWDG4VY4YVLT4H7VCV","json":"https://pith.science/pith/ASSWOMG3RWDG4VY4YVLT4H7VCV.json","graph_json":"https://pith.science/api/pith-number/ASSWOMG3RWDG4VY4YVLT4H7VCV/graph.json","events_json":"https://pith.science/api/pith-number/ASSWOMG3RWDG4VY4YVLT4H7VCV/events.json","paper":"https://pith.science/paper/ASSWOMG3"},"agent_actions":{"view_html":"https://pith.science/pith/ASSWOMG3RWDG4VY4YVLT4H7VCV","download_json":"https://pith.science/pith/ASSWOMG3RWDG4VY4YVLT4H7VCV.json","view_paper":"https://pith.science/paper/ASSWOMG3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.00761&json=true","fetch_graph":"https://pith.science/api/pith-number/ASSWOMG3RWDG4VY4YVLT4H7VCV/graph.json","fetch_events":"https://pith.science/api/pith-number/ASSWOMG3RWDG4VY4YVLT4H7VCV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ASSWOMG3RWDG4VY4YVLT4H7VCV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ASSWOMG3RWDG4VY4YVLT4H7VCV/action/storage_attestation","attest_author":"https://pith.science/pith/ASSWOMG3RWDG4VY4YVLT4H7VCV/action/author_attestation","sign_citation":"https://pith.science/pith/ASSWOMG3RWDG4VY4YVLT4H7VCV/action/citation_signature","submit_replication":"https://pith.science/pith/ASSWOMG3RWDG4VY4YVLT4H7VCV/action/replication_record"}},"created_at":"2026-07-05T10:12:00.269968+00:00","updated_at":"2026-07-05T10:12:00.269968+00:00"}