{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:V6DAUWV6FDP5DZLYDTKTJ5TLM6","short_pith_number":"pith:V6DAUWV6","schema_version":"1.0","canonical_sha256":"af860a5abe28dfd1e5781cd534f66b67965a6e92337d2fcba05c2a62eed0d877","source":{"kind":"arxiv","id":"2406.11813","version":3},"attestation_state":"computed","paper":{"title":"How Do Large Language Models Acquire Factual Knowledge During Pretraining?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Du-Seong Chang, Hoyeon Chang, Jinho Park, Minjoon Seo, Seonghyeon Ye, Sohee Yang, Youngkyung Seo","submitted_at":"2024-06-17T17:54:40Z","abstract_excerpt":"Despite the recent observation that large language models (LLMs) can store substantial factual knowledge, there is a limited understanding of the mechanisms of how they acquire factual knowledge through pretraining. This work addresses this gap by studying how LLMs acquire factual knowledge during pretraining. The findings reveal several important insights into the dynamics of factual knowledge acquisition during pretraining. First, counterintuitively, we observe that pretraining on more data shows no significant improvement in the model's capability to acquire and maintain factual knowledge. "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.11813","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-17T17:54:40Z","cross_cats_sorted":[],"title_canon_sha256":"645bab1cc484854c8dc88b5666d4b6b6831033325acfa5f41056e15be2982321","abstract_canon_sha256":"63f50a086c21909d97b67151aedf737670f727dea2923a13276cc94e69e98c8e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:34:15.005414Z","signature_b64":"muTe/ZbhOm6Zf81cd+H8Oj1GbiIXaF4OXBcpwU9CGGFIvEV61qdUgd2EBVfZrSH6j4SlHb5p3GPy+VSeNZAoBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"af860a5abe28dfd1e5781cd534f66b67965a6e92337d2fcba05c2a62eed0d877","last_reissued_at":"2026-07-05T09:34:15.004890Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:34:15.004890Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"How Do Large Language Models Acquire Factual Knowledge During Pretraining?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Du-Seong Chang, Hoyeon Chang, Jinho Park, Minjoon Seo, Seonghyeon Ye, Sohee Yang, Youngkyung Seo","submitted_at":"2024-06-17T17:54:40Z","abstract_excerpt":"Despite the recent observation that large language models (LLMs) can store substantial factual knowledge, there is a limited understanding of the mechanisms of how they acquire factual knowledge through pretraining. This work addresses this gap by studying how LLMs acquire factual knowledge during pretraining. The findings reveal several important insights into the dynamics of factual knowledge acquisition during pretraining. First, counterintuitively, we observe that pretraining on more data shows no significant improvement in the model's capability to acquire and maintain factual knowledge. "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.11813","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.11813/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.11813","created_at":"2026-07-05T09:34:15.004946+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.11813v3","created_at":"2026-07-05T09:34:15.004946+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.11813","created_at":"2026-07-05T09:34:15.004946+00:00"},{"alias_kind":"pith_short_12","alias_value":"V6DAUWV6FDP5","created_at":"2026-07-05T09:34:15.004946+00:00"},{"alias_kind":"pith_short_16","alias_value":"V6DAUWV6FDP5DZLY","created_at":"2026-07-05T09:34:15.004946+00:00"},{"alias_kind":"pith_short_8","alias_value":"V6DAUWV6","created_at":"2026-07-05T09:34:15.004946+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.01045","citing_title":"Child-directed speech facilitates production, not comprehension, in BabyLMs","ref_index":158,"is_internal_anchor":false},{"citing_arxiv_id":"2412.20760","citing_title":"Attributing Culture-Conditioned Generations to Pretraining Corpora","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/V6DAUWV6FDP5DZLYDTKTJ5TLM6","json":"https://pith.science/pith/V6DAUWV6FDP5DZLYDTKTJ5TLM6.json","graph_json":"https://pith.science/api/pith-number/V6DAUWV6FDP5DZLYDTKTJ5TLM6/graph.json","events_json":"https://pith.science/api/pith-number/V6DAUWV6FDP5DZLYDTKTJ5TLM6/events.json","paper":"https://pith.science/paper/V6DAUWV6"},"agent_actions":{"view_html":"https://pith.science/pith/V6DAUWV6FDP5DZLYDTKTJ5TLM6","download_json":"https://pith.science/pith/V6DAUWV6FDP5DZLYDTKTJ5TLM6.json","view_paper":"https://pith.science/paper/V6DAUWV6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.11813&json=true","fetch_graph":"https://pith.science/api/pith-number/V6DAUWV6FDP5DZLYDTKTJ5TLM6/graph.json","fetch_events":"https://pith.science/api/pith-number/V6DAUWV6FDP5DZLYDTKTJ5TLM6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/V6DAUWV6FDP5DZLYDTKTJ5TLM6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/V6DAUWV6FDP5DZLYDTKTJ5TLM6/action/storage_attestation","attest_author":"https://pith.science/pith/V6DAUWV6FDP5DZLYDTKTJ5TLM6/action/author_attestation","sign_citation":"https://pith.science/pith/V6DAUWV6FDP5DZLYDTKTJ5TLM6/action/citation_signature","submit_replication":"https://pith.science/pith/V6DAUWV6FDP5DZLYDTKTJ5TLM6/action/replication_record"}},"created_at":"2026-07-05T09:34:15.004946+00:00","updated_at":"2026-07-05T09:34:15.004946+00:00"}