{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:U3BH7PLVQTLEWJOL42RITWNSC5","short_pith_number":"pith:U3BH7PLV","schema_version":"1.0","canonical_sha256":"a6c27fbd7584d64b25cbe6a289d9b21778ae1248f7556a4a2c5d407995f21d2a","source":{"kind":"arxiv","id":"2211.08411","version":2},"attestation_state":"computed","paper":{"title":"Large Language Models Struggle to Learn Long-Tail Knowledge","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Adam Roberts, Colin Raffel, Eric Wallace, Haikang Deng, Nikhil Kandpal","submitted_at":"2022-11-15T18:49:27Z","abstract_excerpt":"The Internet contains a wealth of knowledge -- from the birthdays of historical figures to tutorials on how to code -- all of which may be learned by language models. However, while certain pieces of information are ubiquitous on the web, others appear extremely rarely. In this paper, we study the relationship between the knowledge memorized by large language models and the information in pre-training datasets scraped from the web. In particular, we show that a language model's ability to answer a fact-based question relates to how many documents associated with that question were seen during "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2211.08411","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2022-11-15T18:49:27Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"e33e953bed4c08daaa11f963f78fa868c5af44951fd80cd280f8c1501be858b3","abstract_canon_sha256":"f8250f4ca69e19038e8675e021c9f096a44b35a33d0a6865f19b07c4177ee83e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:35:05.571033Z","signature_b64":"G+odSiOUYlIuJ+cyFfsDsXVlV1gKqJ4r5caAghtCDh9ODQ6o2LtlBK6MuQn/5aedlk4/MG7yaAk4sBSl5LFcDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a6c27fbd7584d64b25cbe6a289d9b21778ae1248f7556a4a2c5d407995f21d2a","last_reissued_at":"2026-07-05T06:35:05.570470Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:35:05.570470Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Large Language Models Struggle to Learn Long-Tail Knowledge","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Adam Roberts, Colin Raffel, Eric Wallace, Haikang Deng, Nikhil Kandpal","submitted_at":"2022-11-15T18:49:27Z","abstract_excerpt":"The Internet contains a wealth of knowledge -- from the birthdays of historical figures to tutorials on how to code -- all of which may be learned by language models. However, while certain pieces of information are ubiquitous on the web, others appear extremely rarely. In this paper, we study the relationship between the knowledge memorized by large language models and the information in pre-training datasets scraped from the web. In particular, we show that a language model's ability to answer a fact-based question relates to how many documents associated with that question were seen during "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2211.08411","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2211.08411/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2211.08411","created_at":"2026-07-05T06:35:05.570540+00:00"},{"alias_kind":"arxiv_version","alias_value":"2211.08411v2","created_at":"2026-07-05T06:35:05.570540+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2211.08411","created_at":"2026-07-05T06:35:05.570540+00:00"},{"alias_kind":"pith_short_12","alias_value":"U3BH7PLVQTLE","created_at":"2026-07-05T06:35:05.570540+00:00"},{"alias_kind":"pith_short_16","alias_value":"U3BH7PLVQTLEWJOL","created_at":"2026-07-05T06:35:05.570540+00:00"},{"alias_kind":"pith_short_8","alias_value":"U3BH7PLV","created_at":"2026-07-05T06:35:05.570540+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07874","citing_title":"Safety is Contextual, LLM-Judges Are Not: Navigating the Rigid Priors of Evaluators","ref_index":99,"is_internal_anchor":false},{"citing_arxiv_id":"2502.03916","citing_title":"Experiments with Large Language Models on Retrieval-Augmented Generation for Closed-Source Simulation Software","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15156","citing_title":"MeMo: Memory as a Model","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18732","citing_title":"Predictable Confabulations: Factual Recall by LLMs Scales with Model Size and Topic Frequency","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2305.15717","citing_title":"The False Promise of Imitating Proprietary LLMs","ref_index":143,"is_internal_anchor":false},{"citing_arxiv_id":"2304.01373","citing_title":"Pythia: A Suite for Analyzing Large Language Models Across Training and Scaling","ref_index":197,"is_internal_anchor":false},{"citing_arxiv_id":"2603.18203","citing_title":"How Psychological Learning Paradigms Shaped and Constrained Artificial Intelligence","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15156","citing_title":"MeMo: Memory as a Model","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2310.11511","citing_title":"Self-RAG: Learning to Retrieve, Generate, and Critique through Self-Reflection","ref_index":90,"is_internal_anchor":false},{"citing_arxiv_id":"2310.11511","citing_title":"Self-RAG: Learning to Retrieve, Generate, and Critique through Self-Reflection","ref_index":75,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/U3BH7PLVQTLEWJOL42RITWNSC5","json":"https://pith.science/pith/U3BH7PLVQTLEWJOL42RITWNSC5.json","graph_json":"https://pith.science/api/pith-number/U3BH7PLVQTLEWJOL42RITWNSC5/graph.json","events_json":"https://pith.science/api/pith-number/U3BH7PLVQTLEWJOL42RITWNSC5/events.json","paper":"https://pith.science/paper/U3BH7PLV"},"agent_actions":{"view_html":"https://pith.science/pith/U3BH7PLVQTLEWJOL42RITWNSC5","download_json":"https://pith.science/pith/U3BH7PLVQTLEWJOL42RITWNSC5.json","view_paper":"https://pith.science/paper/U3BH7PLV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2211.08411&json=true","fetch_graph":"https://pith.science/api/pith-number/U3BH7PLVQTLEWJOL42RITWNSC5/graph.json","fetch_events":"https://pith.science/api/pith-number/U3BH7PLVQTLEWJOL42RITWNSC5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/U3BH7PLVQTLEWJOL42RITWNSC5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/U3BH7PLVQTLEWJOL42RITWNSC5/action/storage_attestation","attest_author":"https://pith.science/pith/U3BH7PLVQTLEWJOL42RITWNSC5/action/author_attestation","sign_citation":"https://pith.science/pith/U3BH7PLVQTLEWJOL42RITWNSC5/action/citation_signature","submit_replication":"https://pith.science/pith/U3BH7PLVQTLEWJOL42RITWNSC5/action/replication_record"}},"created_at":"2026-07-05T06:35:05.570540+00:00","updated_at":"2026-07-05T06:35:05.570540+00:00"}