{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:QRPQHNMFV7DEXS7HYR5ILQ5FCZ","short_pith_number":"pith:QRPQHNMF","schema_version":"1.0","canonical_sha256":"845f03b585afc64bcbe7c47a85c3a51678544e91743aead9fbd7d95dd6d1007b","source":{"kind":"arxiv","id":"2002.08056","version":1},"attestation_state":"computed","paper":{"title":"The Geometry of Sign Gradient Descent","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Fabian Pedregosa, Lukas Balles, Nicolas Le Roux","submitted_at":"2020-02-19T08:45:54Z","abstract_excerpt":"Sign-based optimization methods have become popular in machine learning due to their favorable communication cost in distributed optimization and their surprisingly good performance in neural network training. Furthermore, they are closely connected to so-called adaptive gradient methods like Adam. Recent works on signSGD have used a non-standard \"separable smoothness\" assumption, whereas some older works study sign gradient descent as steepest descent with respect to the $\\ell_\\infty$-norm. In this work, we unify these existing results by showing a close connection between separable smoothnes"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2002.08056","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-02-19T08:45:54Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"1f8419284b4b86c830f55d13c99d20daa4adaf237d886c5c916fc8eeb83d8770","abstract_canon_sha256":"d24f1f0339e3a38f173ee3e86e8ac14f8eca34dce3ca3ad4328925239d6f0730"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:42:25.957618Z","signature_b64":"xAyMyZLFhOxdlQD/m7e+4q24C/JEWzGS/mmah0JPKU5pi0hEWh1IK8BmuLeEKlnA8Paps7LJI4AqVpt1Q/eNBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"845f03b585afc64bcbe7c47a85c3a51678544e91743aead9fbd7d95dd6d1007b","last_reissued_at":"2026-07-05T00:42:25.957214Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:42:25.957214Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Geometry of Sign Gradient Descent","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Fabian Pedregosa, Lukas Balles, Nicolas Le Roux","submitted_at":"2020-02-19T08:45:54Z","abstract_excerpt":"Sign-based optimization methods have become popular in machine learning due to their favorable communication cost in distributed optimization and their surprisingly good performance in neural network training. Furthermore, they are closely connected to so-called adaptive gradient methods like Adam. Recent works on signSGD have used a non-standard \"separable smoothness\" assumption, whereas some older works study sign gradient descent as steepest descent with respect to the $\\ell_\\infty$-norm. In this work, we unify these existing results by showing a close connection between separable smoothnes"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2002.08056","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2002.08056/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2002.08056","created_at":"2026-07-05T00:42:25.957278+00:00"},{"alias_kind":"arxiv_version","alias_value":"2002.08056v1","created_at":"2026-07-05T00:42:25.957278+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2002.08056","created_at":"2026-07-05T00:42:25.957278+00:00"},{"alias_kind":"pith_short_12","alias_value":"QRPQHNMFV7DE","created_at":"2026-07-05T00:42:25.957278+00:00"},{"alias_kind":"pith_short_16","alias_value":"QRPQHNMFV7DEXS7H","created_at":"2026-07-05T00:42:25.957278+00:00"},{"alias_kind":"pith_short_8","alias_value":"QRPQHNMF","created_at":"2026-07-05T00:42:25.957278+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23676","citing_title":"Open Problem: Is AdamW Effective Under Heavy-Tailed Noise?","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.14970","citing_title":"Zero-order Parameter-free Optimization for LMO-based Methods: Novel Approach for Efficient Fine-tuning","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02078","citing_title":"Beyond $\\ell_2$-norm and $\\ell_\\infty$-norm: A Curvature-Inspired $\\ell_p$-Norm Scheme for Deep Neural Networks","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2502.07529","citing_title":"Training Deep Learning Models with Norm-Constrained LMOs","ref_index":156,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06654","citing_title":"Optimizer-Model Consistency: Full Finetuning with the Same Optimizer as Pretraining Forgets Less","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06615","citing_title":"When and Why SignSGD Outperforms SGD: A Theoretical Study Based on $\\ell_1$-norm Lower Bounds","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17423","citing_title":"A unified convergence theory for adaptive first-order methods in the nonconvex case, including AdaNorm, full and diagonal AdaGrad, Shampoo and Muo","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QRPQHNMFV7DEXS7HYR5ILQ5FCZ","json":"https://pith.science/pith/QRPQHNMFV7DEXS7HYR5ILQ5FCZ.json","graph_json":"https://pith.science/api/pith-number/QRPQHNMFV7DEXS7HYR5ILQ5FCZ/graph.json","events_json":"https://pith.science/api/pith-number/QRPQHNMFV7DEXS7HYR5ILQ5FCZ/events.json","paper":"https://pith.science/paper/QRPQHNMF"},"agent_actions":{"view_html":"https://pith.science/pith/QRPQHNMFV7DEXS7HYR5ILQ5FCZ","download_json":"https://pith.science/pith/QRPQHNMFV7DEXS7HYR5ILQ5FCZ.json","view_paper":"https://pith.science/paper/QRPQHNMF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2002.08056&json=true","fetch_graph":"https://pith.science/api/pith-number/QRPQHNMFV7DEXS7HYR5ILQ5FCZ/graph.json","fetch_events":"https://pith.science/api/pith-number/QRPQHNMFV7DEXS7HYR5ILQ5FCZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QRPQHNMFV7DEXS7HYR5ILQ5FCZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QRPQHNMFV7DEXS7HYR5ILQ5FCZ/action/storage_attestation","attest_author":"https://pith.science/pith/QRPQHNMFV7DEXS7HYR5ILQ5FCZ/action/author_attestation","sign_citation":"https://pith.science/pith/QRPQHNMFV7DEXS7HYR5ILQ5FCZ/action/citation_signature","submit_replication":"https://pith.science/pith/QRPQHNMFV7DEXS7HYR5ILQ5FCZ/action/replication_record"}},"created_at":"2026-07-05T00:42:25.957278+00:00","updated_at":"2026-07-05T00:42:25.957278+00:00"}