{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:LNBVQ2IFW4TLB3EWGX32KLMKBV","short_pith_number":"pith:LNBVQ2IF","schema_version":"1.0","canonical_sha256":"5b43586905b726b0ec9635f7a52d8a0d4ccbf4e861232735ccec68fca3bef925","source":{"kind":"arxiv","id":"2407.17817","version":1},"attestation_state":"computed","paper":{"title":"Demystifying Verbatim Memorization in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Christopher Potts, Diyi Yang, Jing Huang","submitted_at":"2024-07-25T07:10:31Z","abstract_excerpt":"Large Language Models (LLMs) frequently memorize long sequences verbatim, often with serious legal and privacy implications. Much prior work has studied such verbatim memorization using observational data. To complement such work, we develop a framework to study verbatim memorization in a controlled setting by continuing pre-training from Pythia checkpoints with injected sequences. We find that (1) non-trivial amounts of repetition are necessary for verbatim memorization to happen; (2) later (and presumably better) checkpoints are more likely to verbatim memorize sequences, even for out-of-dis"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.17817","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-07-25T07:10:31Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"38dd77b63106bd27a28f8b094701a5dda7f1bd82372de5bec7f0045355a7591e","abstract_canon_sha256":"4d6b071fcf7486caf6a61a3fd8b83b1fd72052c93e63f52f29db14fe196df691"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:48:22.072808Z","signature_b64":"aj3J97YpuoR0LGw8+lSfujHoNwPnnJXv27rDFdxMXpX2FLyHWegAsQe38TG/vYEE9w30LIr5+583SvpcSYRMDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5b43586905b726b0ec9635f7a52d8a0d4ccbf4e861232735ccec68fca3bef925","last_reissued_at":"2026-07-05T08:48:22.072382Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:48:22.072382Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Demystifying Verbatim Memorization in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Christopher Potts, Diyi Yang, Jing Huang","submitted_at":"2024-07-25T07:10:31Z","abstract_excerpt":"Large Language Models (LLMs) frequently memorize long sequences verbatim, often with serious legal and privacy implications. Much prior work has studied such verbatim memorization using observational data. To complement such work, we develop a framework to study verbatim memorization in a controlled setting by continuing pre-training from Pythia checkpoints with injected sequences. We find that (1) non-trivial amounts of repetition are necessary for verbatim memorization to happen; (2) later (and presumably better) checkpoints are more likely to verbatim memorize sequences, even for out-of-dis"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.17817","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.17817/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.17817","created_at":"2026-07-05T08:48:22.072439+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.17817v1","created_at":"2026-07-05T08:48:22.072439+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.17817","created_at":"2026-07-05T08:48:22.072439+00:00"},{"alias_kind":"pith_short_12","alias_value":"LNBVQ2IFW4TL","created_at":"2026-07-05T08:48:22.072439+00:00"},{"alias_kind":"pith_short_16","alias_value":"LNBVQ2IFW4TLB3EW","created_at":"2026-07-05T08:48:22.072439+00:00"},{"alias_kind":"pith_short_8","alias_value":"LNBVQ2IF","created_at":"2026-07-05T08:48:22.072439+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.23603","citing_title":"SoK: Semantic Privacy in Large Language Models","ref_index":17,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LNBVQ2IFW4TLB3EWGX32KLMKBV","json":"https://pith.science/pith/LNBVQ2IFW4TLB3EWGX32KLMKBV.json","graph_json":"https://pith.science/api/pith-number/LNBVQ2IFW4TLB3EWGX32KLMKBV/graph.json","events_json":"https://pith.science/api/pith-number/LNBVQ2IFW4TLB3EWGX32KLMKBV/events.json","paper":"https://pith.science/paper/LNBVQ2IF"},"agent_actions":{"view_html":"https://pith.science/pith/LNBVQ2IFW4TLB3EWGX32KLMKBV","download_json":"https://pith.science/pith/LNBVQ2IFW4TLB3EWGX32KLMKBV.json","view_paper":"https://pith.science/paper/LNBVQ2IF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.17817&json=true","fetch_graph":"https://pith.science/api/pith-number/LNBVQ2IFW4TLB3EWGX32KLMKBV/graph.json","fetch_events":"https://pith.science/api/pith-number/LNBVQ2IFW4TLB3EWGX32KLMKBV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LNBVQ2IFW4TLB3EWGX32KLMKBV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LNBVQ2IFW4TLB3EWGX32KLMKBV/action/storage_attestation","attest_author":"https://pith.science/pith/LNBVQ2IFW4TLB3EWGX32KLMKBV/action/author_attestation","sign_citation":"https://pith.science/pith/LNBVQ2IFW4TLB3EWGX32KLMKBV/action/citation_signature","submit_replication":"https://pith.science/pith/LNBVQ2IFW4TLB3EWGX32KLMKBV/action/replication_record"}},"created_at":"2026-07-05T08:48:22.072439+00:00","updated_at":"2026-07-05T08:48:22.072439+00:00"}