{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:3SFTKQURR6G3WRWTZ57GHP5RTF","short_pith_number":"pith:3SFTKQUR","schema_version":"1.0","canonical_sha256":"dc8b3542918f8dbb46d3cf7e63bfb19975ad3abf392784fc772582ad2cb6034d","source":{"kind":"arxiv","id":"2408.10682","version":1},"attestation_state":"computed","paper":{"title":"Towards Robust Knowledge Unlearning: An Adversarial Framework for Assessing and Improving Unlearning Robustness in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Hongbang Yuan, Jun Zhao, Kang Liu, Pengfei Cao, Yubo Chen, Zhuoran Jin","submitted_at":"2024-08-20T09:36:04Z","abstract_excerpt":"LLM have achieved success in many fields but still troubled by problematic content in the training corpora. LLM unlearning aims at reducing their influence and avoid undesirable behaviours. However, existing unlearning methods remain vulnerable to adversarial queries and the unlearned knowledge resurfaces after the manually designed attack queries. As part of a red-team effort to proactively assess the vulnerabilities of unlearned models, we design Dynamic Unlearning Attack (DUA), a dynamic and automated framework to attack these models and evaluate their robustness. It optimizes adversarial s"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.10682","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-08-20T09:36:04Z","cross_cats_sorted":["cs.AI","cs.CR","cs.LG"],"title_canon_sha256":"0bae0e96e9a9df6df26b19d088df2ed8b4185724413afd1be10e9d3269a9734d","abstract_canon_sha256":"26c37dc9d572fc04ebb261b1918c8927bb418fd8f85a05ab0a9f1ec22709129d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:57:15.783726Z","signature_b64":"xaSMJ34wWYCzQXkpfuKGViFHeVW8MCaVKgs0Lkynf5SfMkBRYg731SgVImJ3G8wwHSvz38Sg7EvcEX6dW7gOAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dc8b3542918f8dbb46d3cf7e63bfb19975ad3abf392784fc772582ad2cb6034d","last_reissued_at":"2026-07-05T08:57:15.783230Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:57:15.783230Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards Robust Knowledge Unlearning: An Adversarial Framework for Assessing and Improving Unlearning Robustness in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Hongbang Yuan, Jun Zhao, Kang Liu, Pengfei Cao, Yubo Chen, Zhuoran Jin","submitted_at":"2024-08-20T09:36:04Z","abstract_excerpt":"LLM have achieved success in many fields but still troubled by problematic content in the training corpora. LLM unlearning aims at reducing their influence and avoid undesirable behaviours. However, existing unlearning methods remain vulnerable to adversarial queries and the unlearned knowledge resurfaces after the manually designed attack queries. As part of a red-team effort to proactively assess the vulnerabilities of unlearned models, we design Dynamic Unlearning Attack (DUA), a dynamic and automated framework to attack these models and evaluate their robustness. It optimizes adversarial s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.10682","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.10682/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.10682","created_at":"2026-07-05T08:57:15.783286+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.10682v1","created_at":"2026-07-05T08:57:15.783286+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.10682","created_at":"2026-07-05T08:57:15.783286+00:00"},{"alias_kind":"pith_short_12","alias_value":"3SFTKQURR6G3","created_at":"2026-07-05T08:57:15.783286+00:00"},{"alias_kind":"pith_short_16","alias_value":"3SFTKQURR6G3WRWT","created_at":"2026-07-05T08:57:15.783286+00:00"},{"alias_kind":"pith_short_8","alias_value":"3SFTKQUR","created_at":"2026-07-05T08:57:15.783286+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06649","citing_title":"POPS: Recovering Unlearned Multi-Modality Knowledge in MLLMs with Prompt-Optimized Parameter Shaking","ref_index":17,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3SFTKQURR6G3WRWTZ57GHP5RTF","json":"https://pith.science/pith/3SFTKQURR6G3WRWTZ57GHP5RTF.json","graph_json":"https://pith.science/api/pith-number/3SFTKQURR6G3WRWTZ57GHP5RTF/graph.json","events_json":"https://pith.science/api/pith-number/3SFTKQURR6G3WRWTZ57GHP5RTF/events.json","paper":"https://pith.science/paper/3SFTKQUR"},"agent_actions":{"view_html":"https://pith.science/pith/3SFTKQURR6G3WRWTZ57GHP5RTF","download_json":"https://pith.science/pith/3SFTKQURR6G3WRWTZ57GHP5RTF.json","view_paper":"https://pith.science/paper/3SFTKQUR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.10682&json=true","fetch_graph":"https://pith.science/api/pith-number/3SFTKQURR6G3WRWTZ57GHP5RTF/graph.json","fetch_events":"https://pith.science/api/pith-number/3SFTKQURR6G3WRWTZ57GHP5RTF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3SFTKQURR6G3WRWTZ57GHP5RTF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3SFTKQURR6G3WRWTZ57GHP5RTF/action/storage_attestation","attest_author":"https://pith.science/pith/3SFTKQURR6G3WRWTZ57GHP5RTF/action/author_attestation","sign_citation":"https://pith.science/pith/3SFTKQURR6G3WRWTZ57GHP5RTF/action/citation_signature","submit_replication":"https://pith.science/pith/3SFTKQURR6G3WRWTZ57GHP5RTF/action/replication_record"}},"created_at":"2026-07-05T08:57:15.783286+00:00","updated_at":"2026-07-05T08:57:15.783286+00:00"}