{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:WAQA4WLRXSTL4PPWB4SRKZLKPW","short_pith_number":"pith:WAQA4WLR","schema_version":"1.0","canonical_sha256":"b0200e5971bca6be3df60f2515656a7dbe542eb876080f3eaafeb5b0aa76b574","source":{"kind":"arxiv","id":"2002.04839","version":3},"attestation_state":"computed","paper":{"title":"LaProp: Separating Momentum and Adaptivity in Adam","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Liu Ziyin, Masahito Ueda, Zhikang T.Wang","submitted_at":"2020-02-12T08:28:19Z","abstract_excerpt":"We identity a by-far-unrecognized problem of Adam-style optimizers which results from unnecessary coupling between momentum and adaptivity. The coupling leads to instability and divergence when the momentum and adaptivity parameters are mismatched. In this work, we propose a method, Laprop, which decouples momentum and adaptivity in the Adam-style methods. We show that the decoupling leads to greater flexibility in the hyperparameters and allows for a straightforward interpolation between the signed gradient methods and the adaptive gradient methods. We experimentally show that Laprop has cons"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2002.04839","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-02-12T08:28:19Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"e1ddebc38cc7eeca5dcd5bf8fb4aa6774a4a195b0dedb97f20346baa518a7bc4","abstract_canon_sha256":"c5bebdf98d8e8d40b92f0777ac415b5cb9bb15bdf8aace7fd068d838de988331"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:48:32.625147Z","signature_b64":"7pszJpXK8BdqSat084+JhLnObftn6ivhjsEoLjgVQMj5xr5eNhIhQnTgJ4eQcZ2FY36iAYij6NIY1gemE2kyDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b0200e5971bca6be3df60f2515656a7dbe542eb876080f3eaafeb5b0aa76b574","last_reissued_at":"2026-07-05T02:48:32.624753Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:48:32.624753Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LaProp: Separating Momentum and Adaptivity in Adam","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Liu Ziyin, Masahito Ueda, Zhikang T.Wang","submitted_at":"2020-02-12T08:28:19Z","abstract_excerpt":"We identity a by-far-unrecognized problem of Adam-style optimizers which results from unnecessary coupling between momentum and adaptivity. The coupling leads to instability and divergence when the momentum and adaptivity parameters are mismatched. In this work, we propose a method, Laprop, which decouples momentum and adaptivity in the Adam-style methods. We show that the decoupling leads to greater flexibility in the hyperparameters and allows for a straightforward interpolation between the signed gradient methods and the adaptive gradient methods. We experimentally show that Laprop has cons"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2002.04839","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2002.04839/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2002.04839","created_at":"2026-07-05T02:48:32.624809+00:00"},{"alias_kind":"arxiv_version","alias_value":"2002.04839v3","created_at":"2026-07-05T02:48:32.624809+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2002.04839","created_at":"2026-07-05T02:48:32.624809+00:00"},{"alias_kind":"pith_short_12","alias_value":"WAQA4WLRXSTL","created_at":"2026-07-05T02:48:32.624809+00:00"},{"alias_kind":"pith_short_16","alias_value":"WAQA4WLRXSTL4PPW","created_at":"2026-07-05T02:48:32.624809+00:00"},{"alias_kind":"pith_short_8","alias_value":"WAQA4WLR","created_at":"2026-07-05T02:48:32.624809+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.06418","citing_title":"Double Preconditioning (DoPr): Optimization for Test-Time Performance, not Validation Loss","ref_index":160,"is_internal_anchor":false},{"citing_arxiv_id":"2601.12238","citing_title":"On the Provable Suboptimality of Momentum SGD in Nonstationary Stochastic Optimization","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2301.04104","citing_title":"Mastering Diverse Domains through World Models","ref_index":60,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WAQA4WLRXSTL4PPWB4SRKZLKPW","json":"https://pith.science/pith/WAQA4WLRXSTL4PPWB4SRKZLKPW.json","graph_json":"https://pith.science/api/pith-number/WAQA4WLRXSTL4PPWB4SRKZLKPW/graph.json","events_json":"https://pith.science/api/pith-number/WAQA4WLRXSTL4PPWB4SRKZLKPW/events.json","paper":"https://pith.science/paper/WAQA4WLR"},"agent_actions":{"view_html":"https://pith.science/pith/WAQA4WLRXSTL4PPWB4SRKZLKPW","download_json":"https://pith.science/pith/WAQA4WLRXSTL4PPWB4SRKZLKPW.json","view_paper":"https://pith.science/paper/WAQA4WLR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2002.04839&json=true","fetch_graph":"https://pith.science/api/pith-number/WAQA4WLRXSTL4PPWB4SRKZLKPW/graph.json","fetch_events":"https://pith.science/api/pith-number/WAQA4WLRXSTL4PPWB4SRKZLKPW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WAQA4WLRXSTL4PPWB4SRKZLKPW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WAQA4WLRXSTL4PPWB4SRKZLKPW/action/storage_attestation","attest_author":"https://pith.science/pith/WAQA4WLRXSTL4PPWB4SRKZLKPW/action/author_attestation","sign_citation":"https://pith.science/pith/WAQA4WLRXSTL4PPWB4SRKZLKPW/action/citation_signature","submit_replication":"https://pith.science/pith/WAQA4WLRXSTL4PPWB4SRKZLKPW/action/replication_record"}},"created_at":"2026-07-05T02:48:32.624809+00:00","updated_at":"2026-07-05T02:48:32.624809+00:00"}