{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:DDVDT3KW2E53RRZ4IGGZ23GVFL","short_pith_number":"pith:DDVDT3KW","schema_version":"1.0","canonical_sha256":"18ea39ed56d13bb8c73c418d9d6cd52aee24a2cdd54e34cc3e59e071700145ad","source":{"kind":"arxiv","id":"2406.15720","version":1},"attestation_state":"computed","paper":{"title":"Scaling Laws for Fact Memorization of Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Kai Ding, Qinyuan Cheng, Xiaonan Li, Xingyu Lu, Xipeng Qiu, Xuanjing Huang","submitted_at":"2024-06-22T03:32:09Z","abstract_excerpt":"Fact knowledge memorization is crucial for Large Language Models (LLM) to generate factual and reliable responses. However, the behaviors of LLM fact memorization remain under-explored. In this paper, we analyze the scaling laws for LLM's fact knowledge and LLMs' behaviors of memorizing different types of facts. We find that LLMs' fact knowledge capacity has a linear and negative exponential law relationship with model size and training epochs, respectively. Estimated by the built scaling law, memorizing the whole Wikidata's facts requires training an LLM with 1000B non-embed parameters for 10"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.15720","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-06-22T03:32:09Z","cross_cats_sorted":[],"title_canon_sha256":"9c08e4d831204e8d364b530891fb05d84beb4639dbd4d580a874c0813278942c","abstract_canon_sha256":"6683190a6df41802eace53e24a7062941681b08849fd1988b1bc457f27ec9dc6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:35:26.585448Z","signature_b64":"uJ/MuHSoTbROeeEilU1ipf0ztc+D6L7GmsHamsebzOk5F5cF4f8pBZ0Cuf8aFFoHxEF58eKs5FseRLfnJA3tCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"18ea39ed56d13bb8c73c418d9d6cd52aee24a2cdd54e34cc3e59e071700145ad","last_reissued_at":"2026-07-05T08:35:26.585010Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:35:26.585010Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scaling Laws for Fact Memorization of Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Kai Ding, Qinyuan Cheng, Xiaonan Li, Xingyu Lu, Xipeng Qiu, Xuanjing Huang","submitted_at":"2024-06-22T03:32:09Z","abstract_excerpt":"Fact knowledge memorization is crucial for Large Language Models (LLM) to generate factual and reliable responses. However, the behaviors of LLM fact memorization remain under-explored. In this paper, we analyze the scaling laws for LLM's fact knowledge and LLMs' behaviors of memorizing different types of facts. We find that LLMs' fact knowledge capacity has a linear and negative exponential law relationship with model size and training epochs, respectively. Estimated by the built scaling law, memorizing the whole Wikidata's facts requires training an LLM with 1000B non-embed parameters for 10"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.15720","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.15720/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.15720","created_at":"2026-07-05T08:35:26.585066+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.15720v1","created_at":"2026-07-05T08:35:26.585066+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.15720","created_at":"2026-07-05T08:35:26.585066+00:00"},{"alias_kind":"pith_short_12","alias_value":"DDVDT3KW2E53","created_at":"2026-07-05T08:35:26.585066+00:00"},{"alias_kind":"pith_short_16","alias_value":"DDVDT3KW2E53RRZ4","created_at":"2026-07-05T08:35:26.585066+00:00"},{"alias_kind":"pith_short_8","alias_value":"DDVDT3KW","created_at":"2026-07-05T08:35:26.585066+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.18732","citing_title":"Predictable Confabulations: Factual Recall by LLMs Scales with Model Size and Topic Frequency","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DDVDT3KW2E53RRZ4IGGZ23GVFL","json":"https://pith.science/pith/DDVDT3KW2E53RRZ4IGGZ23GVFL.json","graph_json":"https://pith.science/api/pith-number/DDVDT3KW2E53RRZ4IGGZ23GVFL/graph.json","events_json":"https://pith.science/api/pith-number/DDVDT3KW2E53RRZ4IGGZ23GVFL/events.json","paper":"https://pith.science/paper/DDVDT3KW"},"agent_actions":{"view_html":"https://pith.science/pith/DDVDT3KW2E53RRZ4IGGZ23GVFL","download_json":"https://pith.science/pith/DDVDT3KW2E53RRZ4IGGZ23GVFL.json","view_paper":"https://pith.science/paper/DDVDT3KW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.15720&json=true","fetch_graph":"https://pith.science/api/pith-number/DDVDT3KW2E53RRZ4IGGZ23GVFL/graph.json","fetch_events":"https://pith.science/api/pith-number/DDVDT3KW2E53RRZ4IGGZ23GVFL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DDVDT3KW2E53RRZ4IGGZ23GVFL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DDVDT3KW2E53RRZ4IGGZ23GVFL/action/storage_attestation","attest_author":"https://pith.science/pith/DDVDT3KW2E53RRZ4IGGZ23GVFL/action/author_attestation","sign_citation":"https://pith.science/pith/DDVDT3KW2E53RRZ4IGGZ23GVFL/action/citation_signature","submit_replication":"https://pith.science/pith/DDVDT3KW2E53RRZ4IGGZ23GVFL/action/replication_record"}},"created_at":"2026-07-05T08:35:26.585066+00:00","updated_at":"2026-07-05T08:35:26.585066+00:00"}