{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:WEMGAX6T45MAKSKXIRDK5GSFYR","short_pith_number":"pith:WEMGAX6T","schema_version":"1.0","canonical_sha256":"b118605fd3e7580549574446ae9a45c443aad23b5d8e09edb0bf40700aa1ec2c","source":{"kind":"arxiv","id":"2405.18392","version":3},"attestation_state":"computed","paper":{"title":"Scaling Laws and Compute-Optimal Training Beyond Fixed Training Durations","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Alexander H\\\"agele, Atli Kosson, Elie Bakouch, Leandro Von Werra, Loubna Ben Allal, Martin Jaggi","submitted_at":"2024-05-28T17:33:54Z","abstract_excerpt":"Scale has become a main ingredient in obtaining strong machine learning models. As a result, understanding a model's scaling properties is key to effectively designing both the right training setup as well as future generations of architectures. In this work, we argue that scale and training research has been needlessly complex due to reliance on the cosine schedule, which prevents training across different lengths for the same model size. We investigate the training behavior of a direct alternative -- constant learning rate and cooldowns -- and find that it scales predictably and reliably sim"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.18392","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-28T17:33:54Z","cross_cats_sorted":[],"title_canon_sha256":"825d1405377257d6f9a085bfd2a47f158dab060b670170e384f583cfdb196a94","abstract_canon_sha256":"baa577191773a807f46bd48637e14148ecb8f83dd927212c0d3e8cfbf1602e74"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:21:44.780149Z","signature_b64":"pFPdTEXVdCWxzCYg5NDc6LvARBkwUStNeG2pT2duM6qmhc+sjr59baoondvPbD9G9V/72LkqDpmT+7KNaAgUDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b118605fd3e7580549574446ae9a45c443aad23b5d8e09edb0bf40700aa1ec2c","last_reissued_at":"2026-07-05T09:21:44.779662Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:21:44.779662Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scaling Laws and Compute-Optimal Training Beyond Fixed Training Durations","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Alexander H\\\"agele, Atli Kosson, Elie Bakouch, Leandro Von Werra, Loubna Ben Allal, Martin Jaggi","submitted_at":"2024-05-28T17:33:54Z","abstract_excerpt":"Scale has become a main ingredient in obtaining strong machine learning models. As a result, understanding a model's scaling properties is key to effectively designing both the right training setup as well as future generations of architectures. In this work, we argue that scale and training research has been needlessly complex due to reliance on the cosine schedule, which prevents training across different lengths for the same model size. We investigate the training behavior of a direct alternative -- constant learning rate and cooldowns -- and find that it scales predictably and reliably sim"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.18392","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.18392/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.18392","created_at":"2026-07-05T09:21:44.779721+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.18392v3","created_at":"2026-07-05T09:21:44.779721+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.18392","created_at":"2026-07-05T09:21:44.779721+00:00"},{"alias_kind":"pith_short_12","alias_value":"WEMGAX6T45MA","created_at":"2026-07-05T09:21:44.779721+00:00"},{"alias_kind":"pith_short_16","alias_value":"WEMGAX6T45MAKSKX","created_at":"2026-07-05T09:21:44.779721+00:00"},{"alias_kind":"pith_short_8","alias_value":"WEMGAX6T","created_at":"2026-07-05T09:21:44.779721+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25971","citing_title":"Improving Neural Network Training by Decoupling the Magnitude and Direction of Weight Vectors","ref_index":103,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31268","citing_title":"Mellum2 Technical Report","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23061","citing_title":"Anytime Training with Schedule-Free Spectral Optimization","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2409.04777","citing_title":"Optimization Hyper-parameter Laws for Large Language Models","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2412.13663","citing_title":"Smarter, Better, Faster, Longer: A Modern Bidirectional Encoder for Fast, Memory Efficient, and Long Context Finetuning and Inference","ref_index":148,"is_internal_anchor":false},{"citing_arxiv_id":"2412.13663","citing_title":"Smarter, Better, Faster, Longer: A Modern Bidirectional Encoder for Fast, Memory Efficient, and Long Context Finetuning and Inference","ref_index":141,"is_internal_anchor":false},{"citing_arxiv_id":"2510.18900","citing_title":"Foundation Models for Discovery and Exploration in Chemical Space","ref_index":281,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12715","citing_title":"Scaling Laws for Mixture Pretraining Under Data Constraints","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13225","citing_title":"Mix, Don't Tune: Bilingual Pre-Training Outperforms Hyperparameter Search in Data-Constrained Settings","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2502.02737","citing_title":"SmolLM2: When Smol Goes Big -- Data-Centric Training of a Small Language Model","ref_index":177,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08366","citing_title":"Scaling-Aware Data Selection for End-to-End Autonomous Driving Systems","ref_index":20,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WEMGAX6T45MAKSKXIRDK5GSFYR","json":"https://pith.science/pith/WEMGAX6T45MAKSKXIRDK5GSFYR.json","graph_json":"https://pith.science/api/pith-number/WEMGAX6T45MAKSKXIRDK5GSFYR/graph.json","events_json":"https://pith.science/api/pith-number/WEMGAX6T45MAKSKXIRDK5GSFYR/events.json","paper":"https://pith.science/paper/WEMGAX6T"},"agent_actions":{"view_html":"https://pith.science/pith/WEMGAX6T45MAKSKXIRDK5GSFYR","download_json":"https://pith.science/pith/WEMGAX6T45MAKSKXIRDK5GSFYR.json","view_paper":"https://pith.science/paper/WEMGAX6T","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.18392&json=true","fetch_graph":"https://pith.science/api/pith-number/WEMGAX6T45MAKSKXIRDK5GSFYR/graph.json","fetch_events":"https://pith.science/api/pith-number/WEMGAX6T45MAKSKXIRDK5GSFYR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WEMGAX6T45MAKSKXIRDK5GSFYR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WEMGAX6T45MAKSKXIRDK5GSFYR/action/storage_attestation","attest_author":"https://pith.science/pith/WEMGAX6T45MAKSKXIRDK5GSFYR/action/author_attestation","sign_citation":"https://pith.science/pith/WEMGAX6T45MAKSKXIRDK5GSFYR/action/citation_signature","submit_replication":"https://pith.science/pith/WEMGAX6T45MAKSKXIRDK5GSFYR/action/replication_record"}},"created_at":"2026-07-05T09:21:44.779721+00:00","updated_at":"2026-07-05T09:21:44.779721+00:00"}