{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ISCAMB7UZE4NEBEZN56ABTOJCW","short_pith_number":"pith:ISCAMB7U","schema_version":"1.0","canonical_sha256":"44840607f4c938d204996f7c00cdc915afdcb45df30c333859e9e12ebf0b33a7","source":{"kind":"arxiv","id":"2409.18025","version":6},"attestation_state":"computed","paper":{"title":"An Adversarial Perspective on Machine Unlearning for AI Safety","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CR"],"primary_cat":"cs.LG","authors_text":"Boyi Wei, Florian Tram\\`er, Jakub {\\L}ucki, Javier Rando, Peter Henderson, Yangsibo Huang","submitted_at":"2024-09-26T16:32:19Z","abstract_excerpt":"Large language models are finetuned to refuse questions about hazardous knowledge, but these protections can often be bypassed. Unlearning methods aim at completely removing hazardous capabilities from models and make them inaccessible to adversaries. This work challenges the fundamental differences between unlearning and traditional safety post-training from an adversarial perspective. We demonstrate that existing jailbreak methods, previously reported as ineffective against unlearning, can be successful when applied carefully. Furthermore, we develop a variety of adaptive methods that recove"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.18025","kind":"arxiv","version":6},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-09-26T16:32:19Z","cross_cats_sorted":["cs.AI","cs.CL","cs.CR"],"title_canon_sha256":"a5fb18fa35092972a793b1b46dedc157836feecbc68ffe086cac3030b00d4bc3","abstract_canon_sha256":"c530488dc7035df7f3b3ec70ecf5d52f3b93b4f22dff00b359b9778447423289"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:13:13.014016Z","signature_b64":"2OfuXLfCN0786D6iLgiYZFNsik9sY1AkyQOpKZnX21hGUnPLulK+AmspQvVAUG2P3t4it3wUHnBRitL0zGoxCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"44840607f4c938d204996f7c00cdc915afdcb45df30c333859e9e12ebf0b33a7","last_reissued_at":"2026-07-05T11:13:13.013479Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:13:13.013479Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"An Adversarial Perspective on Machine Unlearning for AI Safety","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CR"],"primary_cat":"cs.LG","authors_text":"Boyi Wei, Florian Tram\\`er, Jakub {\\L}ucki, Javier Rando, Peter Henderson, Yangsibo Huang","submitted_at":"2024-09-26T16:32:19Z","abstract_excerpt":"Large language models are finetuned to refuse questions about hazardous knowledge, but these protections can often be bypassed. Unlearning methods aim at completely removing hazardous capabilities from models and make them inaccessible to adversaries. This work challenges the fundamental differences between unlearning and traditional safety post-training from an adversarial perspective. We demonstrate that existing jailbreak methods, previously reported as ineffective against unlearning, can be successful when applied carefully. Furthermore, we develop a variety of adaptive methods that recove"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.18025","kind":"arxiv","version":6},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.18025/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.18025","created_at":"2026-07-05T11:13:13.013542+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.18025v6","created_at":"2026-07-05T11:13:13.013542+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.18025","created_at":"2026-07-05T11:13:13.013542+00:00"},{"alias_kind":"pith_short_12","alias_value":"ISCAMB7UZE4N","created_at":"2026-07-05T11:13:13.013542+00:00"},{"alias_kind":"pith_short_16","alias_value":"ISCAMB7UZE4NEBEZ","created_at":"2026-07-05T11:13:13.013542+00:00"},{"alias_kind":"pith_short_8","alias_value":"ISCAMB7U","created_at":"2026-07-05T11:13:13.013542+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17168","citing_title":"RepSelect: Robust LLM Unlearning via Representation Selectivity","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00105","citing_title":"Visual-Noise Guided In-Context Distillation for Multimodal Large Language Model Unlearning","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2509.22483","citing_title":"OFMU: Optimization-Driven Framework for Machine Unlearning","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2510.00761","citing_title":"Downgrade to Upgrade: Optimizer Simplification Enhances Robustness in LLM Unlearning","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07962","citing_title":"Is your algorithm unlearning or untraining?","ref_index":20,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ISCAMB7UZE4NEBEZN56ABTOJCW","json":"https://pith.science/pith/ISCAMB7UZE4NEBEZN56ABTOJCW.json","graph_json":"https://pith.science/api/pith-number/ISCAMB7UZE4NEBEZN56ABTOJCW/graph.json","events_json":"https://pith.science/api/pith-number/ISCAMB7UZE4NEBEZN56ABTOJCW/events.json","paper":"https://pith.science/paper/ISCAMB7U"},"agent_actions":{"view_html":"https://pith.science/pith/ISCAMB7UZE4NEBEZN56ABTOJCW","download_json":"https://pith.science/pith/ISCAMB7UZE4NEBEZN56ABTOJCW.json","view_paper":"https://pith.science/paper/ISCAMB7U","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.18025&json=true","fetch_graph":"https://pith.science/api/pith-number/ISCAMB7UZE4NEBEZN56ABTOJCW/graph.json","fetch_events":"https://pith.science/api/pith-number/ISCAMB7UZE4NEBEZN56ABTOJCW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ISCAMB7UZE4NEBEZN56ABTOJCW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ISCAMB7UZE4NEBEZN56ABTOJCW/action/storage_attestation","attest_author":"https://pith.science/pith/ISCAMB7UZE4NEBEZN56ABTOJCW/action/author_attestation","sign_citation":"https://pith.science/pith/ISCAMB7UZE4NEBEZN56ABTOJCW/action/citation_signature","submit_replication":"https://pith.science/pith/ISCAMB7UZE4NEBEZN56ABTOJCW/action/replication_record"}},"created_at":"2026-07-05T11:13:13.013542+00:00","updated_at":"2026-07-05T11:13:13.013542+00:00"}