{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:PYUHV3XSCFZGJTPX3IQNEY67C6","short_pith_number":"pith:PYUHV3XS","schema_version":"1.0","canonical_sha256":"7e287aeef2117264cdf7da20d263df17b8b8612c65aa876d6f258d28e512cc03","source":{"kind":"arxiv","id":"2407.14985","version":5},"attestation_state":"computed","paper":{"title":"Generalization v.s. Memorization: Tracing Language Models' Capabilities Back to Pretraining Data","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Alfonso Amayuelas, Alon Albalak, Antonis Antoniades, Kexun Zhang, William Yang Wang, Xinyi Wang, Yanai Elazar","submitted_at":"2024-07-20T21:24:40Z","abstract_excerpt":"The impressive capabilities of large language models (LLMs) have sparked debate over whether these models genuinely generalize to unseen tasks or predominantly rely on memorizing vast amounts of pretraining data. To explore this issue, we introduce an extended concept of memorization, distributional memorization, which measures the correlation between the LLM output probabilities and the pretraining data frequency. To effectively capture task-specific pretraining data frequency, we propose a novel task-gram language model, which is built by counting the co-occurrence of semantically related $n"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.14985","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-07-20T21:24:40Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"35e17b4a797bc1a488d08ec2c9c7e71f4e4165440f083a551fdc0c50daf6d2aa","abstract_canon_sha256":"578629439cf7be2c35681e0c085831687925f6db68ef3de10c38c003e327068a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:22:02.506564Z","signature_b64":"rcBEKIsbIxaDGFDpGuxmrOwhvVwAm0Zc1kbsZb8FV5zsg58P9R2VvBIoMlcCWxvKnztRdXHut/dpkdXwOy1oBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7e287aeef2117264cdf7da20d263df17b8b8612c65aa876d6f258d28e512cc03","last_reissued_at":"2026-07-05T10:22:02.506022Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:22:02.506022Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Generalization v.s. Memorization: Tracing Language Models' Capabilities Back to Pretraining Data","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Alfonso Amayuelas, Alon Albalak, Antonis Antoniades, Kexun Zhang, William Yang Wang, Xinyi Wang, Yanai Elazar","submitted_at":"2024-07-20T21:24:40Z","abstract_excerpt":"The impressive capabilities of large language models (LLMs) have sparked debate over whether these models genuinely generalize to unseen tasks or predominantly rely on memorizing vast amounts of pretraining data. To explore this issue, we introduce an extended concept of memorization, distributional memorization, which measures the correlation between the LLM output probabilities and the pretraining data frequency. To effectively capture task-specific pretraining data frequency, we propose a novel task-gram language model, which is built by counting the co-occurrence of semantically related $n"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.14985","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.14985/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.14985","created_at":"2026-07-05T10:22:02.506084+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.14985v5","created_at":"2026-07-05T10:22:02.506084+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.14985","created_at":"2026-07-05T10:22:02.506084+00:00"},{"alias_kind":"pith_short_12","alias_value":"PYUHV3XSCFZG","created_at":"2026-07-05T10:22:02.506084+00:00"},{"alias_kind":"pith_short_16","alias_value":"PYUHV3XSCFZGJTPX","created_at":"2026-07-05T10:22:02.506084+00:00"},{"alias_kind":"pith_short_8","alias_value":"PYUHV3XS","created_at":"2026-07-05T10:22:02.506084+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.06286","citing_title":"LLMs Can Leak Training Data But Do They Want To? A Propensity-Aware Evaluation of Memorization in LLMs","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2412.20760","citing_title":"Attributing Culture-Conditioned Generations to Pretraining Corpora","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2509.00303","citing_title":"Access Paths for Efficient Ordering with Large Language Models","ref_index":67,"is_internal_anchor":false},{"citing_arxiv_id":"2511.00476","citing_title":"Remembering Unequally: Global and Disciplinary Bias in LLM Reconstruction of Scholarly Coauthor Lists","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2501.17161","citing_title":"SFT Memorizes, RL Generalizes: A Comparative Study of Foundation Model Post-training","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PYUHV3XSCFZGJTPX3IQNEY67C6","json":"https://pith.science/pith/PYUHV3XSCFZGJTPX3IQNEY67C6.json","graph_json":"https://pith.science/api/pith-number/PYUHV3XSCFZGJTPX3IQNEY67C6/graph.json","events_json":"https://pith.science/api/pith-number/PYUHV3XSCFZGJTPX3IQNEY67C6/events.json","paper":"https://pith.science/paper/PYUHV3XS"},"agent_actions":{"view_html":"https://pith.science/pith/PYUHV3XSCFZGJTPX3IQNEY67C6","download_json":"https://pith.science/pith/PYUHV3XSCFZGJTPX3IQNEY67C6.json","view_paper":"https://pith.science/paper/PYUHV3XS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.14985&json=true","fetch_graph":"https://pith.science/api/pith-number/PYUHV3XSCFZGJTPX3IQNEY67C6/graph.json","fetch_events":"https://pith.science/api/pith-number/PYUHV3XSCFZGJTPX3IQNEY67C6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PYUHV3XSCFZGJTPX3IQNEY67C6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PYUHV3XSCFZGJTPX3IQNEY67C6/action/storage_attestation","attest_author":"https://pith.science/pith/PYUHV3XSCFZGJTPX3IQNEY67C6/action/author_attestation","sign_citation":"https://pith.science/pith/PYUHV3XSCFZGJTPX3IQNEY67C6/action/citation_signature","submit_replication":"https://pith.science/pith/PYUHV3XSCFZGJTPX3IQNEY67C6/action/replication_record"}},"created_at":"2026-07-05T10:22:02.506084+00:00","updated_at":"2026-07-05T10:22:02.506084+00:00"}