{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:WOGODOQ6ML3GAK5MYGZWELUPTD","short_pith_number":"pith:WOGODOQ6","schema_version":"1.0","canonical_sha256":"b38ce1ba1e62f6602bacc1b3622e8f98ed4d9267e2b371c82984ca3d4c4e4b55","source":{"kind":"arxiv","id":"2509.07972","version":1},"attestation_state":"computed","paper":{"title":"Theoretical Analysis on how Learning Rate Warmup Accelerates Convergence","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["math.OC"],"primary_cat":"cs.LG","authors_text":"An Kang, Rui Pan, Tong Zhang, Yuxing Liu, Yuze Ge","submitted_at":"2025-09-09T17:56:03Z","abstract_excerpt":"Learning rate warmup is a popular and practical technique in training large-scale deep neural networks. Despite the huge success in practice, the theoretical advantages of this strategy of gradually increasing the learning rate at the beginning of the training process have not been fully understood. To resolve this gap between theory and practice, we first propose a novel family of generalized smoothness assumptions, and validate its applicability both theoretically and empirically. Under the novel smoothness assumption, we study the convergence properties of gradient descent (GD) in both dete"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.07972","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-09-09T17:56:03Z","cross_cats_sorted":["math.OC"],"title_canon_sha256":"a2550ff37db9b288cbbdbecb27b3aa8a003bc5d28fd9ff4e5196c7bdaa97f490","abstract_canon_sha256":"5b503a2b192c6df6f1b7defe3b7e846adf60be34464318c9a14e7dba88865e59"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:07:43.640362Z","signature_b64":"5ZZq8PkNuSeFTqYWEOudD7eq1/d/LB42NrHn8o+FpLmA/SalMByfVsBKOgYWa5XpHOy/TSeus35VvQlMt5TGDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b38ce1ba1e62f6602bacc1b3622e8f98ed4d9267e2b371c82984ca3d4c4e4b55","last_reissued_at":"2026-07-05T12:07:43.639882Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:07:43.639882Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Theoretical Analysis on how Learning Rate Warmup Accelerates Convergence","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["math.OC"],"primary_cat":"cs.LG","authors_text":"An Kang, Rui Pan, Tong Zhang, Yuxing Liu, Yuze Ge","submitted_at":"2025-09-09T17:56:03Z","abstract_excerpt":"Learning rate warmup is a popular and practical technique in training large-scale deep neural networks. Despite the huge success in practice, the theoretical advantages of this strategy of gradually increasing the learning rate at the beginning of the training process have not been fully understood. To resolve this gap between theory and practice, we first propose a novel family of generalized smoothness assumptions, and validate its applicability both theoretically and empirically. Under the novel smoothness assumption, we study the convergence properties of gradient descent (GD) in both dete"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.07972","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.07972/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.07972","created_at":"2026-07-05T12:07:43.639934+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.07972v1","created_at":"2026-07-05T12:07:43.639934+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.07972","created_at":"2026-07-05T12:07:43.639934+00:00"},{"alias_kind":"pith_short_12","alias_value":"WOGODOQ6ML3G","created_at":"2026-07-05T12:07:43.639934+00:00"},{"alias_kind":"pith_short_16","alias_value":"WOGODOQ6ML3GAK5M","created_at":"2026-07-05T12:07:43.639934+00:00"},{"alias_kind":"pith_short_8","alias_value":"WOGODOQ6","created_at":"2026-07-05T12:07:43.639934+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.14800","citing_title":"Avoiding Bias in Clipped SGD for Overparameterized Models under Generalized Smoothness","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26481","citing_title":"A Provably Robust Multi-Jet Framework applied to Active Flow Control of an Airfoil in Weakly Compressible Flow","ref_index":57,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WOGODOQ6ML3GAK5MYGZWELUPTD","json":"https://pith.science/pith/WOGODOQ6ML3GAK5MYGZWELUPTD.json","graph_json":"https://pith.science/api/pith-number/WOGODOQ6ML3GAK5MYGZWELUPTD/graph.json","events_json":"https://pith.science/api/pith-number/WOGODOQ6ML3GAK5MYGZWELUPTD/events.json","paper":"https://pith.science/paper/WOGODOQ6"},"agent_actions":{"view_html":"https://pith.science/pith/WOGODOQ6ML3GAK5MYGZWELUPTD","download_json":"https://pith.science/pith/WOGODOQ6ML3GAK5MYGZWELUPTD.json","view_paper":"https://pith.science/paper/WOGODOQ6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.07972&json=true","fetch_graph":"https://pith.science/api/pith-number/WOGODOQ6ML3GAK5MYGZWELUPTD/graph.json","fetch_events":"https://pith.science/api/pith-number/WOGODOQ6ML3GAK5MYGZWELUPTD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WOGODOQ6ML3GAK5MYGZWELUPTD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WOGODOQ6ML3GAK5MYGZWELUPTD/action/storage_attestation","attest_author":"https://pith.science/pith/WOGODOQ6ML3GAK5MYGZWELUPTD/action/author_attestation","sign_citation":"https://pith.science/pith/WOGODOQ6ML3GAK5MYGZWELUPTD/action/citation_signature","submit_replication":"https://pith.science/pith/WOGODOQ6ML3GAK5MYGZWELUPTD/action/replication_record"}},"created_at":"2026-07-05T12:07:43.639934+00:00","updated_at":"2026-07-05T12:07:43.639934+00:00"}