{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:MNCTPPW4UOKK25Q6UKMN3YYCPV","short_pith_number":"pith:MNCTPPW4","schema_version":"1.0","canonical_sha256":"634537bedca394ad761ea298dde3027d65cc25a81f022a454a2cba353baaf0ca","source":{"kind":"arxiv","id":"2002.09572","version":1},"attestation_state":"computed","paper":{"title":"The Break-Even Point on Optimization Trajectories of Deep Neural Networks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Devansh Arpit, Jacek Tabor, Krzysztof Geras, Kyunghyun Cho, Maciej Szymczak, Stanislav Fort, Stanislaw Jastrzebski","submitted_at":"2020-02-21T22:55:51Z","abstract_excerpt":"The early phase of training of deep neural networks is critical for their final performance. In this work, we study how the hyperparameters of stochastic gradient descent (SGD) used in the early phase of training affect the rest of the optimization trajectory. We argue for the existence of the \"break-even\" point on this trajectory, beyond which the curvature of the loss surface and noise in the gradient are implicitly regularized by SGD. In particular, we demonstrate on multiple classification tasks that using a large learning rate in the initial phase of training reduces the variance of the g"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2002.09572","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-02-21T22:55:51Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"b68d4474d9314f20ea66404a871ad489d19d72fcf5097bd40bbfef1bba8fa82a","abstract_canon_sha256":"9a68fa3fa5b2d03b33fed4e24b6b5b08708d35a2e10009570e8b61d90f0f7d03"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:43:09.159206Z","signature_b64":"jGOWYimYM0u6BcC/PTNmAVbZvel7c5Dz3EycY1yP9Qe6sPzL549ArCQF9QAs2jFk2UQtn+gRoWU4BNBeu61JAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"634537bedca394ad761ea298dde3027d65cc25a81f022a454a2cba353baaf0ca","last_reissued_at":"2026-07-05T00:43:09.158671Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:43:09.158671Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Break-Even Point on Optimization Trajectories of Deep Neural Networks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Devansh Arpit, Jacek Tabor, Krzysztof Geras, Kyunghyun Cho, Maciej Szymczak, Stanislav Fort, Stanislaw Jastrzebski","submitted_at":"2020-02-21T22:55:51Z","abstract_excerpt":"The early phase of training of deep neural networks is critical for their final performance. In this work, we study how the hyperparameters of stochastic gradient descent (SGD) used in the early phase of training affect the rest of the optimization trajectory. We argue for the existence of the \"break-even\" point on this trajectory, beyond which the curvature of the loss surface and noise in the gradient are implicitly regularized by SGD. In particular, we demonstrate on multiple classification tasks that using a large learning rate in the initial phase of training reduces the variance of the g"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2002.09572","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2002.09572/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2002.09572","created_at":"2026-07-05T00:43:09.158737+00:00"},{"alias_kind":"arxiv_version","alias_value":"2002.09572v1","created_at":"2026-07-05T00:43:09.158737+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2002.09572","created_at":"2026-07-05T00:43:09.158737+00:00"},{"alias_kind":"pith_short_12","alias_value":"MNCTPPW4UOKK","created_at":"2026-07-05T00:43:09.158737+00:00"},{"alias_kind":"pith_short_16","alias_value":"MNCTPPW4UOKK25Q6","created_at":"2026-07-05T00:43:09.158737+00:00"},{"alias_kind":"pith_short_8","alias_value":"MNCTPPW4","created_at":"2026-07-05T00:43:09.158737+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.04212","citing_title":"Edge of Stability Selectively Shapes Learning Across the Data Distribution","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16622","citing_title":"Does Weight Decay Enhance Training Stability?","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22778","citing_title":"The Spectral Lifecycle of Transformer Training: Transient Compression Waves, Persistent Spectral Gradients, and the Q/K--V Asymmetry","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21691","citing_title":"There Will Be a Scientific Theory of Deep Learning","ref_index":113,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14108","citing_title":"Momentum Further Constrains Sharpness at the Edge of Stochastic Stability","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19740","citing_title":"Generalization at the Edge of Stability","ref_index":37,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MNCTPPW4UOKK25Q6UKMN3YYCPV","json":"https://pith.science/pith/MNCTPPW4UOKK25Q6UKMN3YYCPV.json","graph_json":"https://pith.science/api/pith-number/MNCTPPW4UOKK25Q6UKMN3YYCPV/graph.json","events_json":"https://pith.science/api/pith-number/MNCTPPW4UOKK25Q6UKMN3YYCPV/events.json","paper":"https://pith.science/paper/MNCTPPW4"},"agent_actions":{"view_html":"https://pith.science/pith/MNCTPPW4UOKK25Q6UKMN3YYCPV","download_json":"https://pith.science/pith/MNCTPPW4UOKK25Q6UKMN3YYCPV.json","view_paper":"https://pith.science/paper/MNCTPPW4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2002.09572&json=true","fetch_graph":"https://pith.science/api/pith-number/MNCTPPW4UOKK25Q6UKMN3YYCPV/graph.json","fetch_events":"https://pith.science/api/pith-number/MNCTPPW4UOKK25Q6UKMN3YYCPV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MNCTPPW4UOKK25Q6UKMN3YYCPV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MNCTPPW4UOKK25Q6UKMN3YYCPV/action/storage_attestation","attest_author":"https://pith.science/pith/MNCTPPW4UOKK25Q6UKMN3YYCPV/action/author_attestation","sign_citation":"https://pith.science/pith/MNCTPPW4UOKK25Q6UKMN3YYCPV/action/citation_signature","submit_replication":"https://pith.science/pith/MNCTPPW4UOKK25Q6UKMN3YYCPV/action/replication_record"}},"created_at":"2026-07-05T00:43:09.158737+00:00","updated_at":"2026-07-05T00:43:09.158737+00:00"}