{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2018:7NHSIOCMI2LWNXLRAVAO3CWQW5","short_pith_number":"pith:7NHSIOCM","schema_version":"1.0","canonical_sha256":"fb4f24384c469766dd710540ed8ad0b7589dc771b4ef9fb9af011f5a9a3dc786","source":{"kind":"arxiv","id":"1806.06763","version":3},"attestation_state":"computed","paper":{"title":"Closing the Generalization Gap of Adaptive Gradient Methods in Training Deep Neural Networks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Dongruo Zhou, Jinghui Chen, Quanquan Gu, Yiqi Tang, Yuan Cao, Ziyan Yang","submitted_at":"2018-06-18T15:17:01Z","abstract_excerpt":"Adaptive gradient methods, which adopt historical gradient information to automatically adjust the learning rate, despite the nice property of fast convergence, have been observed to generalize worse than stochastic gradient descent (SGD) with momentum in training deep neural networks. This leaves how to close the generalization gap of adaptive gradient methods an open problem. In this work, we show that adaptive gradient methods such as Adam, Amsgrad, are sometimes \"over adapted\". We design a new algorithm, called Partially adaptive momentum estimation method, which unifies the Adam/Amsgrad w"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1806.06763","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2018-06-18T15:17:01Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"b0a16ff62cca94fbe3b184fa6755a8df2e58de77dae0b0725bb0a483e7bc96fd","abstract_canon_sha256":"30447ac4c5e1d033c9c8d04eb287eafbff382e93a6e1d073b3b266774d940588"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:12:16.501385Z","signature_b64":"+IsFxqnRdMFePRLn8CWDfzY0tVSCK5Y7T1r2900+iB+Lk83aA+lVUUwhjTVjjB9UHEWX2aVCOWpFHMgeUZPRBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fb4f24384c469766dd710540ed8ad0b7589dc771b4ef9fb9af011f5a9a3dc786","last_reissued_at":"2026-07-05T01:12:16.500913Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:12:16.500913Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Closing the Generalization Gap of Adaptive Gradient Methods in Training Deep Neural Networks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Dongruo Zhou, Jinghui Chen, Quanquan Gu, Yiqi Tang, Yuan Cao, Ziyan Yang","submitted_at":"2018-06-18T15:17:01Z","abstract_excerpt":"Adaptive gradient methods, which adopt historical gradient information to automatically adjust the learning rate, despite the nice property of fast convergence, have been observed to generalize worse than stochastic gradient descent (SGD) with momentum in training deep neural networks. This leaves how to close the generalization gap of adaptive gradient methods an open problem. In this work, we show that adaptive gradient methods such as Adam, Amsgrad, are sometimes \"over adapted\". We design a new algorithm, called Partially adaptive momentum estimation method, which unifies the Adam/Amsgrad w"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1806.06763","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1806.06763/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1806.06763","created_at":"2026-07-05T01:12:16.500969+00:00"},{"alias_kind":"arxiv_version","alias_value":"1806.06763v3","created_at":"2026-07-05T01:12:16.500969+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1806.06763","created_at":"2026-07-05T01:12:16.500969+00:00"},{"alias_kind":"pith_short_12","alias_value":"7NHSIOCMI2LW","created_at":"2026-07-05T01:12:16.500969+00:00"},{"alias_kind":"pith_short_16","alias_value":"7NHSIOCMI2LWNXLR","created_at":"2026-07-05T01:12:16.500969+00:00"},{"alias_kind":"pith_short_8","alias_value":"7NHSIOCM","created_at":"2026-07-05T01:12:16.500969+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09658","citing_title":"Muon Learns More Robust and Transferable Features than Adam","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02317","citing_title":"Anon: Extrapolating Adaptivity Beyond SGD and Adam","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07841","citing_title":"\\mathsf{VISTA}: Decentralized Machine Learning in Adversary Dominated Environments","ref_index":50,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7NHSIOCMI2LWNXLRAVAO3CWQW5","json":"https://pith.science/pith/7NHSIOCMI2LWNXLRAVAO3CWQW5.json","graph_json":"https://pith.science/api/pith-number/7NHSIOCMI2LWNXLRAVAO3CWQW5/graph.json","events_json":"https://pith.science/api/pith-number/7NHSIOCMI2LWNXLRAVAO3CWQW5/events.json","paper":"https://pith.science/paper/7NHSIOCM"},"agent_actions":{"view_html":"https://pith.science/pith/7NHSIOCMI2LWNXLRAVAO3CWQW5","download_json":"https://pith.science/pith/7NHSIOCMI2LWNXLRAVAO3CWQW5.json","view_paper":"https://pith.science/paper/7NHSIOCM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1806.06763&json=true","fetch_graph":"https://pith.science/api/pith-number/7NHSIOCMI2LWNXLRAVAO3CWQW5/graph.json","fetch_events":"https://pith.science/api/pith-number/7NHSIOCMI2LWNXLRAVAO3CWQW5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7NHSIOCMI2LWNXLRAVAO3CWQW5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7NHSIOCMI2LWNXLRAVAO3CWQW5/action/storage_attestation","attest_author":"https://pith.science/pith/7NHSIOCMI2LWNXLRAVAO3CWQW5/action/author_attestation","sign_citation":"https://pith.science/pith/7NHSIOCMI2LWNXLRAVAO3CWQW5/action/citation_signature","submit_replication":"https://pith.science/pith/7NHSIOCMI2LWNXLRAVAO3CWQW5/action/replication_record"}},"created_at":"2026-07-05T01:12:16.500969+00:00","updated_at":"2026-07-05T01:12:16.500969+00:00"}