{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:YQMS7XRYQVPPVG6EYGORCWT4S2","short_pith_number":"pith:YQMS7XRY","schema_version":"1.0","canonical_sha256":"c4192fde38855efa9bc4c19d115a7c9689bfffe842fba2ad3841562bf8638edd","source":{"kind":"arxiv","id":"2405.15523","version":2},"attestation_state":"computed","paper":{"title":"The Mosaic Memory of Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Igor Shilov, Matthieu Meeus, Yves-Alexandre de Montjoye","submitted_at":"2024-05-24T13:05:05Z","abstract_excerpt":"As Large Language Models (LLMs) become widely adopted, understanding how they learn from, and memorize, training data becomes crucial. Memorization in LLMs is widely assumed to only occur as a result of sequences being repeated in the training data. Instead, we show that LLMs memorize by assembling information from similar sequences, a phenomena we call mosaic memory. We show major LLMs to exhibit mosaic memory, with fuzzy duplicates contributing to memorization as much as 0.8 of an exact duplicate and even heavily modified sequences contributing substantially to memorization. Despite models d"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.15523","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-05-24T13:05:05Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"4c778aaab62ff0979971fe4ce7d5c60c42bc668bf883226169db8091ef986ca8","abstract_canon_sha256":"6ba88108747ed09e573150bbbda0206a76673f1780c01ff1771ea690cbce1270"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:03:14.812250Z","signature_b64":"n4KwJEdj6HxzmYn4+gVKHctCIE8PAX7UpRQfXPraMDXF4HV9l6fzTmxqAlac8dXMZXfqWLN9EQDC0FbyV2qmCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c4192fde38855efa9bc4c19d115a7c9689bfffe842fba2ad3841562bf8638edd","last_reissued_at":"2026-07-05T11:03:14.811830Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:03:14.811830Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Mosaic Memory of Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Igor Shilov, Matthieu Meeus, Yves-Alexandre de Montjoye","submitted_at":"2024-05-24T13:05:05Z","abstract_excerpt":"As Large Language Models (LLMs) become widely adopted, understanding how they learn from, and memorize, training data becomes crucial. Memorization in LLMs is widely assumed to only occur as a result of sequences being repeated in the training data. Instead, we show that LLMs memorize by assembling information from similar sequences, a phenomena we call mosaic memory. We show major LLMs to exhibit mosaic memory, with fuzzy duplicates contributing to memorization as much as 0.8 of an exact duplicate and even heavily modified sequences contributing substantially to memorization. Despite models d"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.15523","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.15523/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.15523","created_at":"2026-07-05T11:03:14.811887+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.15523v2","created_at":"2026-07-05T11:03:14.811887+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.15523","created_at":"2026-07-05T11:03:14.811887+00:00"},{"alias_kind":"pith_short_12","alias_value":"YQMS7XRYQVPP","created_at":"2026-07-05T11:03:14.811887+00:00"},{"alias_kind":"pith_short_16","alias_value":"YQMS7XRYQVPPVG6E","created_at":"2026-07-05T11:03:14.811887+00:00"},{"alias_kind":"pith_short_8","alias_value":"YQMS7XRY","created_at":"2026-07-05T11:03:14.811887+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2406.11794","citing_title":"DataComp-LM: In search of the next generation of training sets for language models","ref_index":166,"is_internal_anchor":false},{"citing_arxiv_id":"2512.08875","citing_title":"When Tables Leak: Attacking String Memorization in LLM-Based Tabular Data Generation","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09990","citing_title":"Merlin: Deterministic Byte-Exact Deduplication for Lossless Context Optimization in Large Language Model Inference","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09611","citing_title":"Byte-Exact Deduplication in Retrieval-Augmented Generation: A Three-Regime Empirical Analysis Across Public Benchmarks","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YQMS7XRYQVPPVG6EYGORCWT4S2","json":"https://pith.science/pith/YQMS7XRYQVPPVG6EYGORCWT4S2.json","graph_json":"https://pith.science/api/pith-number/YQMS7XRYQVPPVG6EYGORCWT4S2/graph.json","events_json":"https://pith.science/api/pith-number/YQMS7XRYQVPPVG6EYGORCWT4S2/events.json","paper":"https://pith.science/paper/YQMS7XRY"},"agent_actions":{"view_html":"https://pith.science/pith/YQMS7XRYQVPPVG6EYGORCWT4S2","download_json":"https://pith.science/pith/YQMS7XRYQVPPVG6EYGORCWT4S2.json","view_paper":"https://pith.science/paper/YQMS7XRY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.15523&json=true","fetch_graph":"https://pith.science/api/pith-number/YQMS7XRYQVPPVG6EYGORCWT4S2/graph.json","fetch_events":"https://pith.science/api/pith-number/YQMS7XRYQVPPVG6EYGORCWT4S2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YQMS7XRYQVPPVG6EYGORCWT4S2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YQMS7XRYQVPPVG6EYGORCWT4S2/action/storage_attestation","attest_author":"https://pith.science/pith/YQMS7XRYQVPPVG6EYGORCWT4S2/action/author_attestation","sign_citation":"https://pith.science/pith/YQMS7XRYQVPPVG6EYGORCWT4S2/action/citation_signature","submit_replication":"https://pith.science/pith/YQMS7XRYQVPPVG6EYGORCWT4S2/action/replication_record"}},"created_at":"2026-07-05T11:03:14.811887+00:00","updated_at":"2026-07-05T11:03:14.811887+00:00"}