{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:IVSRHKSHBPPC2XW3BYCSNVZZ6E","short_pith_number":"pith:IVSRHKSH","schema_version":"1.0","canonical_sha256":"456513aa470bde2d5edb0e0526d739f10ae7dfb36f7246cf38824ac0c0b58e13","source":{"kind":"arxiv","id":"2010.05627","version":2},"attestation_state":"computed","paper":{"title":"Towards Theoretically Understanding Why SGD Generalizes Better Than ADAM in Deep Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Caiming Xiong, Chao Ma, Jiashi Feng, Pan Zhou, Steven Hoi, Weinan E","submitted_at":"2020-10-12T12:00:26Z","abstract_excerpt":"It is not clear yet why ADAM-alike adaptive gradient algorithms suffer from worse generalization performance than SGD despite their faster training speed. This work aims to provide understandings on this generalization gap by analyzing their local convergence behaviors. Specifically, we observe the heavy tails of gradient noise in these algorithms. This motivates us to analyze these algorithms through their Levy-driven stochastic differential equations (SDEs) because of the similar convergence behaviors of an algorithm and its SDE. Then we establish the escaping time of these SDEs from a local"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2010.05627","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-10-12T12:00:26Z","cross_cats_sorted":["cs.AI","math.OC","stat.ML"],"title_canon_sha256":"12ad96d58ee5f4c7a035dc1848edfa53b0a0377b35f731cb0f8f69d41767d64a","abstract_canon_sha256":"bf074c4c913c5d0372f6c61576910d2788e7fe3dfeb1660fffafdc1df345b830"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:35:26.543116Z","signature_b64":"1bkbMbdtkWZ4HEs8guPn5klxZXNktVcNI1Ih0ykmy1IjqqghmMw62GNZ4LDoI69vZxella0ZSMqqP9AXBCIEDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"456513aa470bde2d5edb0e0526d739f10ae7dfb36f7246cf38824ac0c0b58e13","last_reissued_at":"2026-07-05T03:35:26.542776Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:35:26.542776Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards Theoretically Understanding Why SGD Generalizes Better Than ADAM in Deep Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Caiming Xiong, Chao Ma, Jiashi Feng, Pan Zhou, Steven Hoi, Weinan E","submitted_at":"2020-10-12T12:00:26Z","abstract_excerpt":"It is not clear yet why ADAM-alike adaptive gradient algorithms suffer from worse generalization performance than SGD despite their faster training speed. This work aims to provide understandings on this generalization gap by analyzing their local convergence behaviors. Specifically, we observe the heavy tails of gradient noise in these algorithms. This motivates us to analyze these algorithms through their Levy-driven stochastic differential equations (SDEs) because of the similar convergence behaviors of an algorithm and its SDE. Then we establish the escaping time of these SDEs from a local"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2010.05627","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2010.05627/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2010.05627","created_at":"2026-07-05T03:35:26.542831+00:00"},{"alias_kind":"arxiv_version","alias_value":"2010.05627v2","created_at":"2026-07-05T03:35:26.542831+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2010.05627","created_at":"2026-07-05T03:35:26.542831+00:00"},{"alias_kind":"pith_short_12","alias_value":"IVSRHKSHBPPC","created_at":"2026-07-05T03:35:26.542831+00:00"},{"alias_kind":"pith_short_16","alias_value":"IVSRHKSHBPPC2XW3","created_at":"2026-07-05T03:35:26.542831+00:00"},{"alias_kind":"pith_short_8","alias_value":"IVSRHKSH","created_at":"2026-07-05T03:35:26.542831+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IVSRHKSHBPPC2XW3BYCSNVZZ6E","json":"https://pith.science/pith/IVSRHKSHBPPC2XW3BYCSNVZZ6E.json","graph_json":"https://pith.science/api/pith-number/IVSRHKSHBPPC2XW3BYCSNVZZ6E/graph.json","events_json":"https://pith.science/api/pith-number/IVSRHKSHBPPC2XW3BYCSNVZZ6E/events.json","paper":"https://pith.science/paper/IVSRHKSH"},"agent_actions":{"view_html":"https://pith.science/pith/IVSRHKSHBPPC2XW3BYCSNVZZ6E","download_json":"https://pith.science/pith/IVSRHKSHBPPC2XW3BYCSNVZZ6E.json","view_paper":"https://pith.science/paper/IVSRHKSH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2010.05627&json=true","fetch_graph":"https://pith.science/api/pith-number/IVSRHKSHBPPC2XW3BYCSNVZZ6E/graph.json","fetch_events":"https://pith.science/api/pith-number/IVSRHKSHBPPC2XW3BYCSNVZZ6E/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IVSRHKSHBPPC2XW3BYCSNVZZ6E/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IVSRHKSHBPPC2XW3BYCSNVZZ6E/action/storage_attestation","attest_author":"https://pith.science/pith/IVSRHKSHBPPC2XW3BYCSNVZZ6E/action/author_attestation","sign_citation":"https://pith.science/pith/IVSRHKSHBPPC2XW3BYCSNVZZ6E/action/citation_signature","submit_replication":"https://pith.science/pith/IVSRHKSHBPPC2XW3BYCSNVZZ6E/action/replication_record"}},"created_at":"2026-07-05T03:35:26.542831+00:00","updated_at":"2026-07-05T03:35:26.542831+00:00"}