{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:G2G67WK27DQTF4BVER6ED5BTLR","short_pith_number":"pith:G2G67WK2","schema_version":"1.0","canonical_sha256":"368defd95af8e132f035247c41f4335c69bb81bbb4d912bc93fba4e0eda21256","source":{"kind":"arxiv","id":"2412.17107","version":3},"attestation_state":"computed","paper":{"title":"Grams: Gradient Descent with Adaptive Momentum Scaling","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.DS","math.OC"],"primary_cat":"cs.LG","authors_text":"Xiaoyu Li, Yang Cao, Zhao Song","submitted_at":"2024-12-22T17:39:32Z","abstract_excerpt":"We introduce $\\mathbf{G}$radient Descent with $\\mathbf{A}$daptive $\\mathbf{M}$omentum $\\mathbf{S}$caling ($\\mathbf{Grams}$), a novel optimization algorithm that decouples the direction and magnitude of parameter updates in deep learning. Unlike traditional optimizers that directly integrate momentum into updates, Grams separates the update direction, derived from current gradients, from momentum, which is used solely for adaptive magnitude scaling. This approach enables Grams to achieve improved loss descent compared to state-of-the-art cautious and momentum-based optimizers. We theoretically "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.17107","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2024-12-22T17:39:32Z","cross_cats_sorted":["cs.AI","cs.DS","math.OC"],"title_canon_sha256":"efb05a396352cb9c582806ea164ee9a9e7fb538de21c660e3e30dde3e99c4667","abstract_canon_sha256":"8eef52efe6a99eb9bd4e1eee23adcb40ed56fbdb8176cd8e27d0524d4ec3c993"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:24:29.155974Z","signature_b64":"+LRmFormqHQ5vW8DJYCmH361Lr+sOdka+2vN26P85oL4wt+1vgcyH4oA382Evr9dB2iYIeQRfYIolpjQuGkBCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"368defd95af8e132f035247c41f4335c69bb81bbb4d912bc93fba4e0eda21256","last_reissued_at":"2026-07-05T10:24:29.155373Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:24:29.155373Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Grams: Gradient Descent with Adaptive Momentum Scaling","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.DS","math.OC"],"primary_cat":"cs.LG","authors_text":"Xiaoyu Li, Yang Cao, Zhao Song","submitted_at":"2024-12-22T17:39:32Z","abstract_excerpt":"We introduce $\\mathbf{G}$radient Descent with $\\mathbf{A}$daptive $\\mathbf{M}$omentum $\\mathbf{S}$caling ($\\mathbf{Grams}$), a novel optimization algorithm that decouples the direction and magnitude of parameter updates in deep learning. Unlike traditional optimizers that directly integrate momentum into updates, Grams separates the update direction, derived from current gradients, from momentum, which is used solely for adaptive magnitude scaling. This approach enables Grams to achieve improved loss descent compared to state-of-the-art cautious and momentum-based optimizers. We theoretically "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.17107","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.17107/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.17107","created_at":"2026-07-05T10:24:29.155446+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.17107v3","created_at":"2026-07-05T10:24:29.155446+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.17107","created_at":"2026-07-05T10:24:29.155446+00:00"},{"alias_kind":"pith_short_12","alias_value":"G2G67WK27DQT","created_at":"2026-07-05T10:24:29.155446+00:00"},{"alias_kind":"pith_short_16","alias_value":"G2G67WK27DQTF4BV","created_at":"2026-07-05T10:24:29.155446+00:00"},{"alias_kind":"pith_short_8","alias_value":"G2G67WK2","created_at":"2026-07-05T10:24:29.155446+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/G2G67WK27DQTF4BVER6ED5BTLR","json":"https://pith.science/pith/G2G67WK27DQTF4BVER6ED5BTLR.json","graph_json":"https://pith.science/api/pith-number/G2G67WK27DQTF4BVER6ED5BTLR/graph.json","events_json":"https://pith.science/api/pith-number/G2G67WK27DQTF4BVER6ED5BTLR/events.json","paper":"https://pith.science/paper/G2G67WK2"},"agent_actions":{"view_html":"https://pith.science/pith/G2G67WK27DQTF4BVER6ED5BTLR","download_json":"https://pith.science/pith/G2G67WK27DQTF4BVER6ED5BTLR.json","view_paper":"https://pith.science/paper/G2G67WK2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.17107&json=true","fetch_graph":"https://pith.science/api/pith-number/G2G67WK27DQTF4BVER6ED5BTLR/graph.json","fetch_events":"https://pith.science/api/pith-number/G2G67WK27DQTF4BVER6ED5BTLR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/G2G67WK27DQTF4BVER6ED5BTLR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/G2G67WK27DQTF4BVER6ED5BTLR/action/storage_attestation","attest_author":"https://pith.science/pith/G2G67WK27DQTF4BVER6ED5BTLR/action/author_attestation","sign_citation":"https://pith.science/pith/G2G67WK27DQTF4BVER6ED5BTLR/action/citation_signature","submit_replication":"https://pith.science/pith/G2G67WK27DQTF4BVER6ED5BTLR/action/replication_record"}},"created_at":"2026-07-05T10:24:29.155446+00:00","updated_at":"2026-07-05T10:24:29.155446+00:00"}