{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:V6YYSJHVI2AZICGD6SBK6JBLZN","short_pith_number":"pith:V6YYSJHV","schema_version":"1.0","canonical_sha256":"afb18924f546819408c3f482af242bcb51fee42a75d5a771c3317d0cf724a4d9","source":{"kind":"arxiv","id":"2306.00204","version":1},"attestation_state":"computed","paper":{"title":"Toward Understanding Why Adam Converges Faster Than SGD for Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Yan Pan, Yuanzhi Li","submitted_at":"2023-05-31T21:49:44Z","abstract_excerpt":"While stochastic gradient descent (SGD) is still the most popular optimization algorithm in deep learning, adaptive algorithms such as Adam have established empirical advantages over SGD in some deep learning applications such as training transformers. However, it remains a question that why Adam converges significantly faster than SGD in these scenarios. In this paper, we propose one explanation of why Adam converges faster than SGD using a new concept directional sharpness. We argue that the performance of optimization algorithms is closely related to the directional sharpness of the update "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.00204","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-05-31T21:49:44Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"953ee2d41fd45142b62d8ae7d6c875338c9d418fd009b86133b75c38d4c8d121","abstract_canon_sha256":"ccf42b2140905b621a0145b38b03425d61b5f4e7b7cab534cb3dedc001d504ea"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:16:23.570423Z","signature_b64":"BCs3X/jbhsx79jkqgC3thQhCkcZYQnsXXXfVKG81HM7aKPXnXtgtcnXwDCgksPngzYnnwlw1Js5mdXhpJWAMCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"afb18924f546819408c3f482af242bcb51fee42a75d5a771c3317d0cf724a4d9","last_reissued_at":"2026-07-05T06:16:23.569900Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:16:23.569900Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Toward Understanding Why Adam Converges Faster Than SGD for Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Yan Pan, Yuanzhi Li","submitted_at":"2023-05-31T21:49:44Z","abstract_excerpt":"While stochastic gradient descent (SGD) is still the most popular optimization algorithm in deep learning, adaptive algorithms such as Adam have established empirical advantages over SGD in some deep learning applications such as training transformers. However, it remains a question that why Adam converges significantly faster than SGD in these scenarios. In this paper, we propose one explanation of why Adam converges faster than SGD using a new concept directional sharpness. We argue that the performance of optimization algorithms is closely related to the directional sharpness of the update "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.00204","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.00204/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.00204","created_at":"2026-07-05T06:16:23.569954+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.00204v1","created_at":"2026-07-05T06:16:23.569954+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.00204","created_at":"2026-07-05T06:16:23.569954+00:00"},{"alias_kind":"pith_short_12","alias_value":"V6YYSJHVI2AZ","created_at":"2026-07-05T06:16:23.569954+00:00"},{"alias_kind":"pith_short_16","alias_value":"V6YYSJHVI2AZICGD","created_at":"2026-07-05T06:16:23.569954+00:00"},{"alias_kind":"pith_short_8","alias_value":"V6YYSJHV","created_at":"2026-07-05T06:16:23.569954+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09658","citing_title":"Muon Learns More Robust and Transferable Features than Adam","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04662","citing_title":"Why Muon Outperforms Adam: A Curvature Perspective","ref_index":174,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00605","citing_title":"Looped Transformers with Layer Normalization Provably Learn the Power Method","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2603.03099","citing_title":"Why Adam Can Beat SGD: Second-Moment Normalization Yields Sharper Tails","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17787","citing_title":"Revisiting the Adam-SGD Gap in LLM Pre-Training: The Role of Large Effective Learning Rates","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2603.03099","citing_title":"Why Adam Can Beat SGD: Second-Moment Normalization Yields Sharper Tails","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14345","citing_title":"Convergence of difference inclusions via a diameter criterion","ref_index":118,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06609","citing_title":"Transformers Efficiently Perform In-Context Logistic Regression via Normalized Gradient Descent","ref_index":37,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/V6YYSJHVI2AZICGD6SBK6JBLZN","json":"https://pith.science/pith/V6YYSJHVI2AZICGD6SBK6JBLZN.json","graph_json":"https://pith.science/api/pith-number/V6YYSJHVI2AZICGD6SBK6JBLZN/graph.json","events_json":"https://pith.science/api/pith-number/V6YYSJHVI2AZICGD6SBK6JBLZN/events.json","paper":"https://pith.science/paper/V6YYSJHV"},"agent_actions":{"view_html":"https://pith.science/pith/V6YYSJHVI2AZICGD6SBK6JBLZN","download_json":"https://pith.science/pith/V6YYSJHVI2AZICGD6SBK6JBLZN.json","view_paper":"https://pith.science/paper/V6YYSJHV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.00204&json=true","fetch_graph":"https://pith.science/api/pith-number/V6YYSJHVI2AZICGD6SBK6JBLZN/graph.json","fetch_events":"https://pith.science/api/pith-number/V6YYSJHVI2AZICGD6SBK6JBLZN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/V6YYSJHVI2AZICGD6SBK6JBLZN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/V6YYSJHVI2AZICGD6SBK6JBLZN/action/storage_attestation","attest_author":"https://pith.science/pith/V6YYSJHVI2AZICGD6SBK6JBLZN/action/author_attestation","sign_citation":"https://pith.science/pith/V6YYSJHVI2AZICGD6SBK6JBLZN/action/citation_signature","submit_replication":"https://pith.science/pith/V6YYSJHVI2AZICGD6SBK6JBLZN/action/replication_record"}},"created_at":"2026-07-05T06:16:23.569954+00:00","updated_at":"2026-07-05T06:16:23.569954+00:00"}