{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:FPF2CDRSZJAY64TPHSC7XJU6IQ","short_pith_number":"pith:FPF2CDRS","schema_version":"1.0","canonical_sha256":"2bcba10e32ca418f726f3c85fba69e441fdcdbafc9d0e901de3aeb05d96663b1","source":{"kind":"arxiv","id":"1908.01878","version":2},"attestation_state":"computed","paper":{"title":"How Does Learning Rate Decay Help Modern Neural Networks?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Jianmin Wang, Kaichao You, Michael I. Jordan, Mingsheng Long","submitted_at":"2019-08-05T21:56:41Z","abstract_excerpt":"Learning rate decay (lrDecay) is a \\emph{de facto} technique for training modern neural networks. It starts with a large learning rate and then decays it multiple times. It is empirically observed to help both optimization and generalization. Common beliefs in how lrDecay works come from the optimization analysis of (Stochastic) Gradient Descent: 1) an initially large learning rate accelerates training or helps the network escape spurious local minima; 2) decaying the learning rate helps the network converge to a local minimum and avoid oscillation. Despite the popularity of these common belie"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1908.01878","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-08-05T21:56:41Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"1b6dfff18c34737b8fdbb6de1796a93b201cd1ed520930b0aac55819ae03ff9f","abstract_canon_sha256":"9cd16416e30938bf8cd9ba3c90f65429bea29737199c4fa033e21a61c3b0e52a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:07:24.394384Z","signature_b64":"RF8kzA+5a7iVPufReeJrGa4TSanA745/nuXa0v88cxcuK5KJYw+XJUBAmz39QZgXlNy5OoT7MVK9s3aHYgfhBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2bcba10e32ca418f726f3c85fba69e441fdcdbafc9d0e901de3aeb05d96663b1","last_reissued_at":"2026-07-05T00:07:24.393855Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:07:24.393855Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"How Does Learning Rate Decay Help Modern Neural Networks?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Jianmin Wang, Kaichao You, Michael I. Jordan, Mingsheng Long","submitted_at":"2019-08-05T21:56:41Z","abstract_excerpt":"Learning rate decay (lrDecay) is a \\emph{de facto} technique for training modern neural networks. It starts with a large learning rate and then decays it multiple times. It is empirically observed to help both optimization and generalization. Common beliefs in how lrDecay works come from the optimization analysis of (Stochastic) Gradient Descent: 1) an initially large learning rate accelerates training or helps the network escape spurious local minima; 2) decaying the learning rate helps the network converge to a local minimum and avoid oscillation. Despite the popularity of these common belie"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1908.01878","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1908.01878/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1908.01878","created_at":"2026-07-05T00:07:24.393920+00:00"},{"alias_kind":"arxiv_version","alias_value":"1908.01878v2","created_at":"2026-07-05T00:07:24.393920+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1908.01878","created_at":"2026-07-05T00:07:24.393920+00:00"},{"alias_kind":"pith_short_12","alias_value":"FPF2CDRSZJAY","created_at":"2026-07-05T00:07:24.393920+00:00"},{"alias_kind":"pith_short_16","alias_value":"FPF2CDRSZJAY64TP","created_at":"2026-07-05T00:07:24.393920+00:00"},{"alias_kind":"pith_short_8","alias_value":"FPF2CDRS","created_at":"2026-07-05T00:07:24.393920+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.23061","citing_title":"Anytime Training with Schedule-Free Spectral Optimization","ref_index":87,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23362","citing_title":"UniAda: Universal Adaptive Multi-objective Adversarial Attack for End-to-End Autonomous Driving Systems","ref_index":48,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FPF2CDRSZJAY64TPHSC7XJU6IQ","json":"https://pith.science/pith/FPF2CDRSZJAY64TPHSC7XJU6IQ.json","graph_json":"https://pith.science/api/pith-number/FPF2CDRSZJAY64TPHSC7XJU6IQ/graph.json","events_json":"https://pith.science/api/pith-number/FPF2CDRSZJAY64TPHSC7XJU6IQ/events.json","paper":"https://pith.science/paper/FPF2CDRS"},"agent_actions":{"view_html":"https://pith.science/pith/FPF2CDRSZJAY64TPHSC7XJU6IQ","download_json":"https://pith.science/pith/FPF2CDRSZJAY64TPHSC7XJU6IQ.json","view_paper":"https://pith.science/paper/FPF2CDRS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1908.01878&json=true","fetch_graph":"https://pith.science/api/pith-number/FPF2CDRSZJAY64TPHSC7XJU6IQ/graph.json","fetch_events":"https://pith.science/api/pith-number/FPF2CDRSZJAY64TPHSC7XJU6IQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FPF2CDRSZJAY64TPHSC7XJU6IQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FPF2CDRSZJAY64TPHSC7XJU6IQ/action/storage_attestation","attest_author":"https://pith.science/pith/FPF2CDRSZJAY64TPHSC7XJU6IQ/action/author_attestation","sign_citation":"https://pith.science/pith/FPF2CDRSZJAY64TPHSC7XJU6IQ/action/citation_signature","submit_replication":"https://pith.science/pith/FPF2CDRSZJAY64TPHSC7XJU6IQ/action/replication_record"}},"created_at":"2026-07-05T00:07:24.393920+00:00","updated_at":"2026-07-05T00:07:24.393920+00:00"}