{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:VO7EWMSKXLHOWZR2QKU2HKJ7V4","short_pith_number":"pith:VO7EWMSK","schema_version":"1.0","canonical_sha256":"abbe4b324abaceeb663a82a9a3a93faf18b72db09a6f8a2680bafaa1dceda1dc","source":{"kind":"arxiv","id":"2405.03097","version":1},"attestation_state":"computed","paper":{"title":"To Each (Textual Sequence) Its Own: Improving Memorized-Data Unlearning in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"George-Octavian Barbulescu, Peter Triantafillou","submitted_at":"2024-05-06T01:21:50Z","abstract_excerpt":"LLMs have been found to memorize training textual sequences and regurgitate verbatim said sequences during text generation time. This fact is known to be the cause of privacy and related (e.g., copyright) problems. Unlearning in LLMs then takes the form of devising new algorithms that will properly deal with these side-effects of memorized data, while not hurting the model's utility. We offer a fresh perspective towards this goal, namely, that each textual sequence to be forgotten should be treated differently when being unlearned based on its degree of memorization within the LLM. We contribu"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.03097","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-06T01:21:50Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"befc3f72941ff29211514cb3c20574556e38bc9bc36e70467d5a71ca17e37164","abstract_canon_sha256":"82d40716f10056fe7f94ccb5205a2fb9cf5f00c65725d9c933474301d265bb40"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:15:58.419784Z","signature_b64":"KLgrGJ4K9BmmxMKAPhcCvpkBljjIGh8BhRdATiJ0W9TidfBNt9ou77rUbuRuwdngNerXKKd3wqDRapav6Mr5Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"abbe4b324abaceeb663a82a9a3a93faf18b72db09a6f8a2680bafaa1dceda1dc","last_reissued_at":"2026-07-05T08:15:58.419352Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:15:58.419352Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"To Each (Textual Sequence) Its Own: Improving Memorized-Data Unlearning in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"George-Octavian Barbulescu, Peter Triantafillou","submitted_at":"2024-05-06T01:21:50Z","abstract_excerpt":"LLMs have been found to memorize training textual sequences and regurgitate verbatim said sequences during text generation time. This fact is known to be the cause of privacy and related (e.g., copyright) problems. Unlearning in LLMs then takes the form of devising new algorithms that will properly deal with these side-effects of memorized data, while not hurting the model's utility. We offer a fresh perspective towards this goal, namely, that each textual sequence to be forgotten should be treated differently when being unlearned based on its degree of memorization within the LLM. We contribu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.03097","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.03097/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.03097","created_at":"2026-07-05T08:15:58.419416+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.03097v1","created_at":"2026-07-05T08:15:58.419416+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.03097","created_at":"2026-07-05T08:15:58.419416+00:00"},{"alias_kind":"pith_short_12","alias_value":"VO7EWMSKXLHO","created_at":"2026-07-05T08:15:58.419416+00:00"},{"alias_kind":"pith_short_16","alias_value":"VO7EWMSKXLHOWZR2","created_at":"2026-07-05T08:15:58.419416+00:00"},{"alias_kind":"pith_short_8","alias_value":"VO7EWMSK","created_at":"2026-07-05T08:15:58.419416+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.02453","citing_title":"Initialization is Half the Battle: Generating Diverse Images from a Guidance Potential Posterior","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2506.20941","citing_title":"Revisiting the Past: Data Unlearning with Model State History","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07962","citing_title":"Is your algorithm unlearning or untraining?","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13777","citing_title":"From Anchors to Supervision: Memory-Graph Guided Corpus-Free Unlearning for Large Language Models","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VO7EWMSKXLHOWZR2QKU2HKJ7V4","json":"https://pith.science/pith/VO7EWMSKXLHOWZR2QKU2HKJ7V4.json","graph_json":"https://pith.science/api/pith-number/VO7EWMSKXLHOWZR2QKU2HKJ7V4/graph.json","events_json":"https://pith.science/api/pith-number/VO7EWMSKXLHOWZR2QKU2HKJ7V4/events.json","paper":"https://pith.science/paper/VO7EWMSK"},"agent_actions":{"view_html":"https://pith.science/pith/VO7EWMSKXLHOWZR2QKU2HKJ7V4","download_json":"https://pith.science/pith/VO7EWMSKXLHOWZR2QKU2HKJ7V4.json","view_paper":"https://pith.science/paper/VO7EWMSK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.03097&json=true","fetch_graph":"https://pith.science/api/pith-number/VO7EWMSKXLHOWZR2QKU2HKJ7V4/graph.json","fetch_events":"https://pith.science/api/pith-number/VO7EWMSKXLHOWZR2QKU2HKJ7V4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VO7EWMSKXLHOWZR2QKU2HKJ7V4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VO7EWMSKXLHOWZR2QKU2HKJ7V4/action/storage_attestation","attest_author":"https://pith.science/pith/VO7EWMSKXLHOWZR2QKU2HKJ7V4/action/author_attestation","sign_citation":"https://pith.science/pith/VO7EWMSKXLHOWZR2QKU2HKJ7V4/action/citation_signature","submit_replication":"https://pith.science/pith/VO7EWMSKXLHOWZR2QKU2HKJ7V4/action/replication_record"}},"created_at":"2026-07-05T08:15:58.419416+00:00","updated_at":"2026-07-05T08:15:58.419416+00:00"}