{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:S252APSGOFMB5QZ6XNA2UKKMWN","short_pith_number":"pith:S252APSG","schema_version":"1.0","canonical_sha256":"96bba03e4671581ec33ebb41aa294cb35ff01d3b03cea9a4845809d239dfd3f8","source":{"kind":"arxiv","id":"2405.14578","version":5},"attestation_state":"computed","paper":{"title":"Surge Phenomenon in Optimal Learning Rate and Batch Size Scaling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Bin Cui, Chengjun Liu, Dian Jiao, Di Wang, Hailin Zhang, Hao Wu, Jinbao Xue, Penghao Zhao, Shuaipeng Li, Weiyan Wang, Xingwu Sun, Yangyu Tao, Zheng Fang","submitted_at":"2024-05-23T13:52:36Z","abstract_excerpt":"In current deep learning tasks, Adam style optimizers such as Adam, Adagrad, RMSProp, Adafactor, and Lion have been widely used as alternatives to SGD style optimizers. These optimizers typically update model parameters using the sign of gradients, resulting in more stable convergence curves. The learning rate and the batch size are the most critical hyperparameters for optimizers, which require careful tuning to enable effective convergence. Previous research has shown that the optimal learning rate increases linearly or follows similar rules with batch size for SGD style optimizers. However,"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.14578","kind":"arxiv","version":5},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-23T13:52:36Z","cross_cats_sorted":[],"title_canon_sha256":"04da36bf2fac23244d2d966e548e488ad1362fefd092f9a9ea5a477467aa97a4","abstract_canon_sha256":"53bf62ce0a0ee9601f38a9033a570beaf26620e91d4fffdaf32af3f5a7937096"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:27:00.802120Z","signature_b64":"MPIaeopA2IeD0yf+QEL1aftWKjynosPfKalj9y+k/Nw+IkNzlLKVaOTsVwcMeNFJc0F1I7Rt7yMtbKhJP9MKCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"96bba03e4671581ec33ebb41aa294cb35ff01d3b03cea9a4845809d239dfd3f8","last_reissued_at":"2026-07-05T09:27:00.801690Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:27:00.801690Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Surge Phenomenon in Optimal Learning Rate and Batch Size Scaling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Bin Cui, Chengjun Liu, Dian Jiao, Di Wang, Hailin Zhang, Hao Wu, Jinbao Xue, Penghao Zhao, Shuaipeng Li, Weiyan Wang, Xingwu Sun, Yangyu Tao, Zheng Fang","submitted_at":"2024-05-23T13:52:36Z","abstract_excerpt":"In current deep learning tasks, Adam style optimizers such as Adam, Adagrad, RMSProp, Adafactor, and Lion have been widely used as alternatives to SGD style optimizers. These optimizers typically update model parameters using the sign of gradients, resulting in more stable convergence curves. The learning rate and the batch size are the most critical hyperparameters for optimizers, which require careful tuning to enable effective convergence. Previous research has shown that the optimal learning rate increases linearly or follows similar rules with batch size for SGD style optimizers. However,"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.14578","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.14578/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.14578","created_at":"2026-07-05T09:27:00.801749+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.14578v5","created_at":"2026-07-05T09:27:00.801749+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.14578","created_at":"2026-07-05T09:27:00.801749+00:00"},{"alias_kind":"pith_short_12","alias_value":"S252APSGOFMB","created_at":"2026-07-05T09:27:00.801749+00:00"},{"alias_kind":"pith_short_16","alias_value":"S252APSGOFMB5QZ6","created_at":"2026-07-05T09:27:00.801749+00:00"},{"alias_kind":"pith_short_8","alias_value":"S252APSG","created_at":"2026-07-05T09:27:00.801749+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/S252APSGOFMB5QZ6XNA2UKKMWN","json":"https://pith.science/pith/S252APSGOFMB5QZ6XNA2UKKMWN.json","graph_json":"https://pith.science/api/pith-number/S252APSGOFMB5QZ6XNA2UKKMWN/graph.json","events_json":"https://pith.science/api/pith-number/S252APSGOFMB5QZ6XNA2UKKMWN/events.json","paper":"https://pith.science/paper/S252APSG"},"agent_actions":{"view_html":"https://pith.science/pith/S252APSGOFMB5QZ6XNA2UKKMWN","download_json":"https://pith.science/pith/S252APSGOFMB5QZ6XNA2UKKMWN.json","view_paper":"https://pith.science/paper/S252APSG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.14578&json=true","fetch_graph":"https://pith.science/api/pith-number/S252APSGOFMB5QZ6XNA2UKKMWN/graph.json","fetch_events":"https://pith.science/api/pith-number/S252APSGOFMB5QZ6XNA2UKKMWN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/S252APSGOFMB5QZ6XNA2UKKMWN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/S252APSGOFMB5QZ6XNA2UKKMWN/action/storage_attestation","attest_author":"https://pith.science/pith/S252APSGOFMB5QZ6XNA2UKKMWN/action/author_attestation","sign_citation":"https://pith.science/pith/S252APSGOFMB5QZ6XNA2UKKMWN/action/citation_signature","submit_replication":"https://pith.science/pith/S252APSGOFMB5QZ6XNA2UKKMWN/action/replication_record"}},"created_at":"2026-07-05T09:27:00.801749+00:00","updated_at":"2026-07-05T09:27:00.801749+00:00"}