{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:CCXZ5FRGHQAGEH3FIF62MDHEZX","short_pith_number":"pith:CCXZ5FRG","schema_version":"1.0","canonical_sha256":"10af9e96263c00621f65417da60ce4cdd92b557fbaac298115782176dd53f4bc","source":{"kind":"arxiv","id":"2406.19146","version":4},"attestation_state":"computed","paper":{"title":"Resolving Discrepancies in Compute-Optimal Scaling of Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Jenia Jitsev, Ludwig Schmidt, Mitchell Wortsman, Tomer Porian, Yair Carmon","submitted_at":"2024-06-27T13:02:43Z","abstract_excerpt":"Kaplan et al. and Hoffmann et al. developed influential scaling laws for the optimal model size as a function of the compute budget, but these laws yield substantially different predictions. We explain the discrepancy by reproducing the Kaplan scaling law on two datasets (OpenWebText2 and RefinedWeb) and identifying three factors causing the difference: last layer computational cost, warmup duration, and scale-dependent optimizer tuning. With these factors corrected, we obtain excellent agreement with the Hoffmann et al. (i.e., \"Chinchilla\") scaling law. Counter to a hypothesis of Hoffmann et "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.19146","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-06-27T13:02:43Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"ef61a420d977a767c2a26251932c53d8360eb028575ca597756d4ab0cba5f9a9","abstract_canon_sha256":"7427e5c1593a96552d423a26d8779d79d385309c5ce80184f8a7a94a9fbad487"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:02:47.844502Z","signature_b64":"d4Qc/mKM7nVqihMkVHHLrIqiPwIB6EwA+wrvtGwp4Qn8si/i2ZFzLWDnfQ5a1DcU7pl6XI9CHvtmWNgEurAPAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"10af9e96263c00621f65417da60ce4cdd92b557fbaac298115782176dd53f4bc","last_reissued_at":"2026-07-05T10:02:47.844049Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:02:47.844049Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Resolving Discrepancies in Compute-Optimal Scaling of Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Jenia Jitsev, Ludwig Schmidt, Mitchell Wortsman, Tomer Porian, Yair Carmon","submitted_at":"2024-06-27T13:02:43Z","abstract_excerpt":"Kaplan et al. and Hoffmann et al. developed influential scaling laws for the optimal model size as a function of the compute budget, but these laws yield substantially different predictions. We explain the discrepancy by reproducing the Kaplan scaling law on two datasets (OpenWebText2 and RefinedWeb) and identifying three factors causing the difference: last layer computational cost, warmup duration, and scale-dependent optimizer tuning. With these factors corrected, we obtain excellent agreement with the Hoffmann et al. (i.e., \"Chinchilla\") scaling law. Counter to a hypothesis of Hoffmann et "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.19146","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.19146/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.19146","created_at":"2026-07-05T10:02:47.844105+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.19146v4","created_at":"2026-07-05T10:02:47.844105+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.19146","created_at":"2026-07-05T10:02:47.844105+00:00"},{"alias_kind":"pith_short_12","alias_value":"CCXZ5FRGHQAG","created_at":"2026-07-05T10:02:47.844105+00:00"},{"alias_kind":"pith_short_16","alias_value":"CCXZ5FRGHQAGEH3F","created_at":"2026-07-05T10:02:47.844105+00:00"},{"alias_kind":"pith_short_8","alias_value":"CCXZ5FRG","created_at":"2026-07-05T10:02:47.844105+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.29158","citing_title":"On the Nonlinearity of Learning Rate Scaling for LLM Training","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29448","citing_title":"How Much Is a Dataset Worth? Scaling Laws, the Vendi Score, and Matrix Spectral Functions","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2502.12120","citing_title":"LLMs on the Line: Data Determines Loss-to-Loss Scaling Laws","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2510.13786","citing_title":"The Art of Scaling Reinforcement Learning Compute for LLMs","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12715","citing_title":"Scaling Laws for Mixture Pretraining Under Data Constraints","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13225","citing_title":"Mix, Don't Tune: Bilingual Pre-Training Outperforms Hyperparameter Search in Data-Constrained Settings","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CCXZ5FRGHQAGEH3FIF62MDHEZX","json":"https://pith.science/pith/CCXZ5FRGHQAGEH3FIF62MDHEZX.json","graph_json":"https://pith.science/api/pith-number/CCXZ5FRGHQAGEH3FIF62MDHEZX/graph.json","events_json":"https://pith.science/api/pith-number/CCXZ5FRGHQAGEH3FIF62MDHEZX/events.json","paper":"https://pith.science/paper/CCXZ5FRG"},"agent_actions":{"view_html":"https://pith.science/pith/CCXZ5FRGHQAGEH3FIF62MDHEZX","download_json":"https://pith.science/pith/CCXZ5FRGHQAGEH3FIF62MDHEZX.json","view_paper":"https://pith.science/paper/CCXZ5FRG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.19146&json=true","fetch_graph":"https://pith.science/api/pith-number/CCXZ5FRGHQAGEH3FIF62MDHEZX/graph.json","fetch_events":"https://pith.science/api/pith-number/CCXZ5FRGHQAGEH3FIF62MDHEZX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CCXZ5FRGHQAGEH3FIF62MDHEZX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CCXZ5FRGHQAGEH3FIF62MDHEZX/action/storage_attestation","attest_author":"https://pith.science/pith/CCXZ5FRGHQAGEH3FIF62MDHEZX/action/author_attestation","sign_citation":"https://pith.science/pith/CCXZ5FRGHQAGEH3FIF62MDHEZX/action/citation_signature","submit_replication":"https://pith.science/pith/CCXZ5FRGHQAGEH3FIF62MDHEZX/action/replication_record"}},"created_at":"2026-07-05T10:02:47.844105+00:00","updated_at":"2026-07-05T10:02:47.844105+00:00"}