{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:CMFQ6UBXLS4KCZYCZOC4ALQU46","short_pith_number":"pith:CMFQ6UBX","schema_version":"1.0","canonical_sha256":"130b0f50375cb8a16702cb85c02e14e78655ac29270c367064f3781007392fcd","source":{"kind":"arxiv","id":"2208.06677","version":5},"attestation_state":"computed","paper":{"title":"Adan: Adaptive Nesterov Momentum Algorithm for Faster Optimizing Deep Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC"],"primary_cat":"cs.LG","authors_text":"Huan Li, Pan Zhou, Shuicheng Yan, Xingyu Xie, Zhouchen Lin","submitted_at":"2022-08-13T16:04:39Z","abstract_excerpt":"In deep learning, different kinds of deep networks typically need different optimizers, which have to be chosen after multiple trials, making the training process inefficient. To relieve this issue and consistently improve the model training speed across deep networks, we propose the ADAptive Nesterov momentum algorithm, Adan for short. Adan first reformulates the vanilla Nesterov acceleration to develop a new Nesterov momentum estimation (NME) method, which avoids the extra overhead of computing gradient at the extrapolation point. Then, Adan adopts NME to estimate the gradient's first- and s"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2208.06677","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-08-13T16:04:39Z","cross_cats_sorted":["math.OC"],"title_canon_sha256":"bacf8eefa35b6d80fc84b172cf3cfdceeb6d3e4f1da15bad82f29427151bf795","abstract_canon_sha256":"fa1eec7a19a75346b3a671caeed12b838e6c8bbc78d072488c82fb52726441f8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:41:26.246369Z","signature_b64":"LRwCYO/HVuorUZ4KlIFP0vJcY01J3pEE5v52gMKwqJmWdiewUFuFsszGWU4ABWyW1LdUG6NJwRW91ghemeqfAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"130b0f50375cb8a16702cb85c02e14e78655ac29270c367064f3781007392fcd","last_reissued_at":"2026-07-05T09:41:26.245810Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:41:26.245810Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Adan: Adaptive Nesterov Momentum Algorithm for Faster Optimizing Deep Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC"],"primary_cat":"cs.LG","authors_text":"Huan Li, Pan Zhou, Shuicheng Yan, Xingyu Xie, Zhouchen Lin","submitted_at":"2022-08-13T16:04:39Z","abstract_excerpt":"In deep learning, different kinds of deep networks typically need different optimizers, which have to be chosen after multiple trials, making the training process inefficient. To relieve this issue and consistently improve the model training speed across deep networks, we propose the ADAptive Nesterov momentum algorithm, Adan for short. Adan first reformulates the vanilla Nesterov acceleration to develop a new Nesterov momentum estimation (NME) method, which avoids the extra overhead of computing gradient at the extrapolation point. Then, Adan adopts NME to estimate the gradient's first- and s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2208.06677","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2208.06677/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2208.06677","created_at":"2026-07-05T09:41:26.245876+00:00"},{"alias_kind":"arxiv_version","alias_value":"2208.06677v5","created_at":"2026-07-05T09:41:26.245876+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2208.06677","created_at":"2026-07-05T09:41:26.245876+00:00"},{"alias_kind":"pith_short_12","alias_value":"CMFQ6UBXLS4K","created_at":"2026-07-05T09:41:26.245876+00:00"},{"alias_kind":"pith_short_16","alias_value":"CMFQ6UBXLS4KCZYC","created_at":"2026-07-05T09:41:26.245876+00:00"},{"alias_kind":"pith_short_8","alias_value":"CMFQ6UBX","created_at":"2026-07-05T09:41:26.245876+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24507","citing_title":"Uncovering Latent Structures in Robust Pulse Sequences: A Model-Based Reinforcement Learning Approach for Adaptable Quantum Control","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2504.10013","citing_title":"Training LLMs on HPC Systems: Best Practices from the OpenGPT-X Project","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2402.17177","citing_title":"Sora: A Review on Background, Technology, Limitations, and Opportunities of Large Vision Models","ref_index":57,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CMFQ6UBXLS4KCZYCZOC4ALQU46","json":"https://pith.science/pith/CMFQ6UBXLS4KCZYCZOC4ALQU46.json","graph_json":"https://pith.science/api/pith-number/CMFQ6UBXLS4KCZYCZOC4ALQU46/graph.json","events_json":"https://pith.science/api/pith-number/CMFQ6UBXLS4KCZYCZOC4ALQU46/events.json","paper":"https://pith.science/paper/CMFQ6UBX"},"agent_actions":{"view_html":"https://pith.science/pith/CMFQ6UBXLS4KCZYCZOC4ALQU46","download_json":"https://pith.science/pith/CMFQ6UBXLS4KCZYCZOC4ALQU46.json","view_paper":"https://pith.science/paper/CMFQ6UBX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2208.06677&json=true","fetch_graph":"https://pith.science/api/pith-number/CMFQ6UBXLS4KCZYCZOC4ALQU46/graph.json","fetch_events":"https://pith.science/api/pith-number/CMFQ6UBXLS4KCZYCZOC4ALQU46/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CMFQ6UBXLS4KCZYCZOC4ALQU46/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CMFQ6UBXLS4KCZYCZOC4ALQU46/action/storage_attestation","attest_author":"https://pith.science/pith/CMFQ6UBXLS4KCZYCZOC4ALQU46/action/author_attestation","sign_citation":"https://pith.science/pith/CMFQ6UBXLS4KCZYCZOC4ALQU46/action/citation_signature","submit_replication":"https://pith.science/pith/CMFQ6UBXLS4KCZYCZOC4ALQU46/action/replication_record"}},"created_at":"2026-07-05T09:41:26.245876+00:00","updated_at":"2026-07-05T09:41:26.245876+00:00"}