{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:EHC42OIH4S3IRN7M46FQCTMN6Z","short_pith_number":"pith:EHC42OIH","schema_version":"1.0","canonical_sha256":"21c5cd3907e4b688b7ece78b014d8df64138f49acacbfe1e55b3045ab0ea2104","source":{"kind":"arxiv","id":"2410.05838","version":2},"attestation_state":"computed","paper":{"title":"Time Transfer: On Optimal Learning Rate and Batch Size In The Infinite Data Limit","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Jan Ebert, Jiangtao Wang, Oleg Filatov, Stefan Kesselheim","submitted_at":"2024-10-08T09:06:34Z","abstract_excerpt":"One of the main challenges in optimal scaling of large language models (LLMs) is the prohibitive cost of hyperparameter tuning, particularly learning rate $\\eta$ and batch size $B$. While techniques like $\\mu$P (Yang et al., 2022) provide scaling rules for optimal $\\eta$ transfer in the infinite model size limit, the optimal scaling behavior in the infinite data size limit remains unknown. We fill in this gap by observing for the first time an intricate dependence of optimal $\\eta$ scaling on the pretraining token budget $T$, $B$ and its relation to the critical batch size $B_\\mathrm{crit}$, w"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.05838","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-08T09:06:34Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"492a820dc01631f3ac1f05b8c46c4e4ec8e3a0e42cd4817defb2837b1bb97f67","abstract_canon_sha256":"1c93520964e4141c87f53d4bd91c345c37ec5aa8026415fc21817a94a54887a5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:59:00.302105Z","signature_b64":"LaFWxGUgbSfNR9k/HEG5XGlu8/R2JGj63XGTIfjTadtwdHdPdhUxO/4lcePpMazbY4NU2B4tk2PqnLm+BAVLAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"21c5cd3907e4b688b7ece78b014d8df64138f49acacbfe1e55b3045ab0ea2104","last_reissued_at":"2026-07-05T09:59:00.301639Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:59:00.301639Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Time Transfer: On Optimal Learning Rate and Batch Size In The Infinite Data Limit","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Jan Ebert, Jiangtao Wang, Oleg Filatov, Stefan Kesselheim","submitted_at":"2024-10-08T09:06:34Z","abstract_excerpt":"One of the main challenges in optimal scaling of large language models (LLMs) is the prohibitive cost of hyperparameter tuning, particularly learning rate $\\eta$ and batch size $B$. While techniques like $\\mu$P (Yang et al., 2022) provide scaling rules for optimal $\\eta$ transfer in the infinite model size limit, the optimal scaling behavior in the infinite data size limit remains unknown. We fill in this gap by observing for the first time an intricate dependence of optimal $\\eta$ scaling on the pretraining token budget $T$, $B$ and its relation to the critical batch size $B_\\mathrm{crit}$, w"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.05838","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.05838/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.05838","created_at":"2026-07-05T09:59:00.301698+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.05838v2","created_at":"2026-07-05T09:59:00.301698+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.05838","created_at":"2026-07-05T09:59:00.301698+00:00"},{"alias_kind":"pith_short_12","alias_value":"EHC42OIH4S3I","created_at":"2026-07-05T09:59:00.301698+00:00"},{"alias_kind":"pith_short_16","alias_value":"EHC42OIH4S3IRN7M","created_at":"2026-07-05T09:59:00.301698+00:00"},{"alias_kind":"pith_short_8","alias_value":"EHC42OIH","created_at":"2026-07-05T09:59:00.301698+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01487","citing_title":"How to Allocate Your Tokens? Scaling Laws with Training Steps and Batch Size","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26687","citing_title":"COPUS: Co-adaptive Parallelism and Batch Size Selection in Large Language Model Training","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EHC42OIH4S3IRN7M46FQCTMN6Z","json":"https://pith.science/pith/EHC42OIH4S3IRN7M46FQCTMN6Z.json","graph_json":"https://pith.science/api/pith-number/EHC42OIH4S3IRN7M46FQCTMN6Z/graph.json","events_json":"https://pith.science/api/pith-number/EHC42OIH4S3IRN7M46FQCTMN6Z/events.json","paper":"https://pith.science/paper/EHC42OIH"},"agent_actions":{"view_html":"https://pith.science/pith/EHC42OIH4S3IRN7M46FQCTMN6Z","download_json":"https://pith.science/pith/EHC42OIH4S3IRN7M46FQCTMN6Z.json","view_paper":"https://pith.science/paper/EHC42OIH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.05838&json=true","fetch_graph":"https://pith.science/api/pith-number/EHC42OIH4S3IRN7M46FQCTMN6Z/graph.json","fetch_events":"https://pith.science/api/pith-number/EHC42OIH4S3IRN7M46FQCTMN6Z/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EHC42OIH4S3IRN7M46FQCTMN6Z/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EHC42OIH4S3IRN7M46FQCTMN6Z/action/storage_attestation","attest_author":"https://pith.science/pith/EHC42OIH4S3IRN7M46FQCTMN6Z/action/author_attestation","sign_citation":"https://pith.science/pith/EHC42OIH4S3IRN7M46FQCTMN6Z/action/citation_signature","submit_replication":"https://pith.science/pith/EHC42OIH4S3IRN7M46FQCTMN6Z/action/replication_record"}},"created_at":"2026-07-05T09:59:00.301698+00:00","updated_at":"2026-07-05T09:59:00.301698+00:00"}