{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:TDVPYQ7V7RR4XPAS7JQCSTQDEJ","short_pith_number":"pith:TDVPYQ7V","schema_version":"1.0","canonical_sha256":"98eafc43f5fc63cbbc12fa60294e032264df4d4346dfa331e58d563da45aa2f1","source":{"kind":"arxiv","id":"2402.16835","version":1},"attestation_state":"computed","paper":{"title":"Eight Methods to Evaluate Robust Unlearning in LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Aengus Lynch, Aidan Ewart, Dylan Hadfield-Menell, Phillip Guo, Stephen Casper","submitted_at":"2024-02-26T18:57:37Z","abstract_excerpt":"Machine unlearning can be useful for removing harmful capabilities and memorized text from large language models (LLMs), but there are not yet standardized methods for rigorously evaluating it. In this paper, we first survey techniques and limitations of existing unlearning evaluations. Second, we apply a comprehensive set of tests for the robustness and competitiveness of unlearning in the \"Who's Harry Potter\" (WHP) model from Eldan and Russinovich (2023). While WHP's unlearning generalizes well when evaluated with the \"Familiarity\" metric from Eldan and Russinovich, we find i) higher-than-ba"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.16835","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-02-26T18:57:37Z","cross_cats_sorted":[],"title_canon_sha256":"ec437d75a1b27e97e43d0db3706ccf7c3d6964da1194d09e91a62ee16eff2790","abstract_canon_sha256":"8b24d15b2a9f65738c227cda06234570719bcc5b47f4dbbb9e7d281b97bda16c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:49:26.597241Z","signature_b64":"5j+fUgFUceviExZRRY/mPOZQ1iw+XIXoXBIxRfIQ53tc/vD1fwwgr7CKC/b9HfSxxeOsFIutXfBP9tD1LrsSBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"98eafc43f5fc63cbbc12fa60294e032264df4d4346dfa331e58d563da45aa2f1","last_reissued_at":"2026-07-05T07:49:26.596798Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:49:26.596798Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Eight Methods to Evaluate Robust Unlearning in LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Aengus Lynch, Aidan Ewart, Dylan Hadfield-Menell, Phillip Guo, Stephen Casper","submitted_at":"2024-02-26T18:57:37Z","abstract_excerpt":"Machine unlearning can be useful for removing harmful capabilities and memorized text from large language models (LLMs), but there are not yet standardized methods for rigorously evaluating it. In this paper, we first survey techniques and limitations of existing unlearning evaluations. Second, we apply a comprehensive set of tests for the robustness and competitiveness of unlearning in the \"Who's Harry Potter\" (WHP) model from Eldan and Russinovich (2023). While WHP's unlearning generalizes well when evaluated with the \"Familiarity\" metric from Eldan and Russinovich, we find i) higher-than-ba"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.16835","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.16835/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.16835","created_at":"2026-07-05T07:49:26.596857+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.16835v1","created_at":"2026-07-05T07:49:26.596857+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.16835","created_at":"2026-07-05T07:49:26.596857+00:00"},{"alias_kind":"pith_short_12","alias_value":"TDVPYQ7V7RR4","created_at":"2026-07-05T07:49:26.596857+00:00"},{"alias_kind":"pith_short_16","alias_value":"TDVPYQ7V7RR4XPAS","created_at":"2026-07-05T07:49:26.596857+00:00"},{"alias_kind":"pith_short_8","alias_value":"TDVPYQ7V","created_at":"2026-07-05T07:49:26.596857+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":20,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17539","citing_title":"Reinforcing Dual-Path Reasoning in Spatial Vision Language Models","ref_index":103,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17168","citing_title":"RepSelect: Robust LLM Unlearning via Representation Selectivity","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03291","citing_title":"Multilingual Unlearning in LLMs: Transfer, Dynamics, and Reversibility","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27379","citing_title":"Position: The Term \"Machine Unlearning\" Is Overused in LLMs","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24614","citing_title":"Measuring the Depth of LLM Unlearning via Activation Patching","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2501.19202","citing_title":"Improving LLM Unlearning Robustness via Random Perturbations","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2505.16831","citing_title":"Unlearning Isn't Deletion: Investigating Reversibility of Machine Unlearning in LLMs","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18891","citing_title":"Auditing Reasoning-Trace Memorization Claims after Unlearning with Head-Conditioned Canaries","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16776","citing_title":"Distinguishable Deletion: Unifying Knowledge Erasure and Refusal for Large Language Model Unlearning","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2510.00761","citing_title":"Downgrade to Upgrade: Optimizer Simplification Enhances Robustness in LLM Unlearning","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11685","citing_title":"Robust LLM Unlearning Against Relearning Attacks: The Minor Components in Representations Matter","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10777","citing_title":"Locking Pretrained Weights via Deep Low-Rank Residual Distillation","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09391","citing_title":"Efficient Unlearning through Maximizing Relearning Convergence Delay","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10403","citing_title":"Latent Instruction Representation Alignment: defending against jailbreaks, backdoors and undesired knowledge in LLMs","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07962","citing_title":"Is your algorithm unlearning or untraining?","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02196","citing_title":"DurableUn: Quantization-Induced Recovery Attacks in Machine Unlearning","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13438","citing_title":"WIN-U: Woodbury-Informed Newton-Unlearning as a retain-free Machine Unlearning Framework","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17396","citing_title":"Representation-Guided Parameter-Efficient LLM Unlearning","ref_index":142,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02196","citing_title":"DurableUn: Quantization-Induced Recovery Attacks in Machine Unlearning","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01699","citing_title":"Probe-Geometry Alignment: Erasing the Cross-Sequence Memorization Signature Below Chance","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TDVPYQ7V7RR4XPAS7JQCSTQDEJ","json":"https://pith.science/pith/TDVPYQ7V7RR4XPAS7JQCSTQDEJ.json","graph_json":"https://pith.science/api/pith-number/TDVPYQ7V7RR4XPAS7JQCSTQDEJ/graph.json","events_json":"https://pith.science/api/pith-number/TDVPYQ7V7RR4XPAS7JQCSTQDEJ/events.json","paper":"https://pith.science/paper/TDVPYQ7V"},"agent_actions":{"view_html":"https://pith.science/pith/TDVPYQ7V7RR4XPAS7JQCSTQDEJ","download_json":"https://pith.science/pith/TDVPYQ7V7RR4XPAS7JQCSTQDEJ.json","view_paper":"https://pith.science/paper/TDVPYQ7V","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.16835&json=true","fetch_graph":"https://pith.science/api/pith-number/TDVPYQ7V7RR4XPAS7JQCSTQDEJ/graph.json","fetch_events":"https://pith.science/api/pith-number/TDVPYQ7V7RR4XPAS7JQCSTQDEJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TDVPYQ7V7RR4XPAS7JQCSTQDEJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TDVPYQ7V7RR4XPAS7JQCSTQDEJ/action/storage_attestation","attest_author":"https://pith.science/pith/TDVPYQ7V7RR4XPAS7JQCSTQDEJ/action/author_attestation","sign_citation":"https://pith.science/pith/TDVPYQ7V7RR4XPAS7JQCSTQDEJ/action/citation_signature","submit_replication":"https://pith.science/pith/TDVPYQ7V7RR4XPAS7JQCSTQDEJ/action/replication_record"}},"created_at":"2026-07-05T07:49:26.596857+00:00","updated_at":"2026-07-05T07:49:26.596857+00:00"}