{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:WXZVI4V2IUP6ELAT6F3F2RTVH6","short_pith_number":"pith:WXZVI4V2","schema_version":"1.0","canonical_sha256":"b5f35472ba451fe22c13f1765d46753fbf1d99dca422760385fccdf2b7e27784","source":{"kind":"arxiv","id":"2505.24832","version":3},"attestation_state":"computed","paper":{"title":"How much do language models memorize?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alexander M. Rush, Chawin Sitawarin, Chuan Guo, G. Edward Suh, John X. Morris, Kamalika Chaudhuri, Narine Kokhlikyan, Saeed Mahloujifar","submitted_at":"2025-05-30T17:34:03Z","abstract_excerpt":"We propose a new method for estimating how much a model knows about a datapoint and use it to measure the capacity of modern language models. Prior studies of language model memorization have struggled to disentangle memorization from generalization. We formally separate memorization into two components: unintended memorization, the information a model contains about a specific dataset, and generalization, the information a model contains about the true data-generation process. When we completely eliminate generalization, we can compute the total memorization, which provides an estimate of mod"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.24832","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-30T17:34:03Z","cross_cats_sorted":[],"title_canon_sha256":"8969aba3fbbe132a80cbafc42ebf0d4e99391987a48849647fef45ba4d99804c","abstract_canon_sha256":"bc4a9903a996f65d446d40088c32aa8de68b925250becd95fcc80ed701185193"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:23:29.000819Z","signature_b64":"dq3NAQHnvv/kh0fLHM97R8AqAarh8ekUDN1Ye0KvsJGZGWwTbE+sfOvvkKo9baF+UNG6FxYPABt2gRMZVwT8BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b5f35472ba451fe22c13f1765d46753fbf1d99dca422760385fccdf2b7e27784","last_reissued_at":"2026-07-05T11:23:29.000310Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:23:29.000310Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"How much do language models memorize?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alexander M. Rush, Chawin Sitawarin, Chuan Guo, G. Edward Suh, John X. Morris, Kamalika Chaudhuri, Narine Kokhlikyan, Saeed Mahloujifar","submitted_at":"2025-05-30T17:34:03Z","abstract_excerpt":"We propose a new method for estimating how much a model knows about a datapoint and use it to measure the capacity of modern language models. Prior studies of language model memorization have struggled to disentangle memorization from generalization. We formally separate memorization into two components: unintended memorization, the information a model contains about a specific dataset, and generalization, the information a model contains about the true data-generation process. When we completely eliminate generalization, we can compute the total memorization, which provides an estimate of mod"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.24832","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.24832/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.24832","created_at":"2026-07-05T11:23:29.000370+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.24832v3","created_at":"2026-07-05T11:23:29.000370+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.24832","created_at":"2026-07-05T11:23:29.000370+00:00"},{"alias_kind":"pith_short_12","alias_value":"WXZVI4V2IUP6","created_at":"2026-07-05T11:23:29.000370+00:00"},{"alias_kind":"pith_short_16","alias_value":"WXZVI4V2IUP6ELAT","created_at":"2026-07-05T11:23:29.000370+00:00"},{"alias_kind":"pith_short_8","alias_value":"WXZVI4V2","created_at":"2026-07-05T11:23:29.000370+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":19,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08393","citing_title":"Towards Mechanistically Understanding Why Memorized Knowledge Fails to Generalize in Large Language Model Finetuning","ref_index":19,"is_internal_anchor":true},{"citing_arxiv_id":"2606.17164","citing_title":"PromptMN: Pseudo Prompting Language","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12764","citing_title":"Detecting Functional Memorization in Code Language Models","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02920","citing_title":"Fast Unlearning at Scale via Margin Self-Correction","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14659","citing_title":"Slower Generalization, Faster Memorization: A Sweet Spot in Algorithmic Learning","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26097","citing_title":"Forgetting in Language Models: Capacity, Optimization, and Self-Generated Replay","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27564","citing_title":"The Future of Facts: Tracing the Factual Generation-Verification Gap","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27446","citing_title":"Causal Connections: Leveraging Multilingual Fine-Tuning for Financial QA@FinCausal 2026","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27316","citing_title":"LLM-Based Examination of Eligibility Criteria from Securities Prospectuses at the German Central Bank","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2503.16771","citing_title":"Enabling Global, Human-Centered Explanations for LLMs:From Tokens to Interpretable Code and Test Generation","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2509.00303","citing_title":"Access Paths for Efficient Ordering with Large Language Models","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2510.26745","citing_title":"Deep sequence models tend to memorize geometrically; it is unclear why","ref_index":121,"is_internal_anchor":false},{"citing_arxiv_id":"2602.12005","citing_title":"LaCy: What Small Language Models Can and Should Learn is Not Just a Question of Loss","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2510.17934","citing_title":"AtlasKV: Augmenting LLMs with Billion-Scale Knowledge Graphs in 20GB VRAM","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09724","citing_title":"Model Capacity Determines Grokking through Competing Memorisation and Generalisation Speeds","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03759","citing_title":"Before Forgetting, Learn to Remember: Revisiting Foundational Learning Failures in LVLM Unlearning Benchmarks","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03547","citing_title":"Erase Persona, Forget Lore: Benchmarking Multimodal Copyright Unlearning in Large Vision Language Models","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24827","citing_title":"Incompressible Knowledge Probes: Estimating Black-Box LLM Parameter Counts via Factual Capacity","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08519","citing_title":"Cram Less to Fit More: Training Data Pruning Improves Memorization of Facts","ref_index":63,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WXZVI4V2IUP6ELAT6F3F2RTVH6","json":"https://pith.science/pith/WXZVI4V2IUP6ELAT6F3F2RTVH6.json","graph_json":"https://pith.science/api/pith-number/WXZVI4V2IUP6ELAT6F3F2RTVH6/graph.json","events_json":"https://pith.science/api/pith-number/WXZVI4V2IUP6ELAT6F3F2RTVH6/events.json","paper":"https://pith.science/paper/WXZVI4V2"},"agent_actions":{"view_html":"https://pith.science/pith/WXZVI4V2IUP6ELAT6F3F2RTVH6","download_json":"https://pith.science/pith/WXZVI4V2IUP6ELAT6F3F2RTVH6.json","view_paper":"https://pith.science/paper/WXZVI4V2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.24832&json=true","fetch_graph":"https://pith.science/api/pith-number/WXZVI4V2IUP6ELAT6F3F2RTVH6/graph.json","fetch_events":"https://pith.science/api/pith-number/WXZVI4V2IUP6ELAT6F3F2RTVH6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WXZVI4V2IUP6ELAT6F3F2RTVH6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WXZVI4V2IUP6ELAT6F3F2RTVH6/action/storage_attestation","attest_author":"https://pith.science/pith/WXZVI4V2IUP6ELAT6F3F2RTVH6/action/author_attestation","sign_citation":"https://pith.science/pith/WXZVI4V2IUP6ELAT6F3F2RTVH6/action/citation_signature","submit_replication":"https://pith.science/pith/WXZVI4V2IUP6ELAT6F3F2RTVH6/action/replication_record"}},"created_at":"2026-07-05T11:23:29.000370+00:00","updated_at":"2026-07-05T11:23:29.000370+00:00"}