{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:RL4CLPXQ5LWOLTG4VPEOTYIDUQ","short_pith_number":"pith:RL4CLPXQ","schema_version":"1.0","canonical_sha256":"8af825bef0eaece5ccdcabc8e9e103a42e00c37c6f6d5a41b9541338d352b0f5","source":{"kind":"arxiv","id":"2002.09018","version":2},"attestation_state":"computed","paper":{"title":"Scalable Second Order Optimization for Deep Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Kevin Regan, Rohan Anil, Tomer Koren, Vineet Gupta, Yoram Singer","submitted_at":"2020-02-20T20:51:33Z","abstract_excerpt":"Optimization in machine learning, both theoretical and applied, is presently dominated by first-order gradient methods such as stochastic gradient descent. Second-order optimization methods, that involve second derivatives and/or second order statistics of the data, are far less prevalent despite strong theoretical properties, due to their prohibitive computation, memory and communication costs. In an attempt to bridge this gap between theoretical and practical optimization, we present a scalable implementation of a second-order preconditioned method (concretely, a variant of full-matrix Adagr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2002.09018","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-02-20T20:51:33Z","cross_cats_sorted":["math.OC","stat.ML"],"title_canon_sha256":"547c3e5e61bc53c6af78549bd8f951430946c1505224aeb22b53427f287d9a8d","abstract_canon_sha256":"09a38a527a240f6c0b304d56ce454ab484555ae7edbebdd6887f73ac48928768"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:20:27.751896Z","signature_b64":"jPthofBx/KXIkLHkqoPmT3IZCBAdEGjXNysJ/J1HzNu+HpLazP0Hc4XXTqI6Xl3aUdC5bavjdnOGONS1jYjUBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8af825bef0eaece5ccdcabc8e9e103a42e00c37c6f6d5a41b9541338d352b0f5","last_reissued_at":"2026-07-05T02:20:27.751420Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:20:27.751420Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scalable Second Order Optimization for Deep Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Kevin Regan, Rohan Anil, Tomer Koren, Vineet Gupta, Yoram Singer","submitted_at":"2020-02-20T20:51:33Z","abstract_excerpt":"Optimization in machine learning, both theoretical and applied, is presently dominated by first-order gradient methods such as stochastic gradient descent. Second-order optimization methods, that involve second derivatives and/or second order statistics of the data, are far less prevalent despite strong theoretical properties, due to their prohibitive computation, memory and communication costs. In an attempt to bridge this gap between theoretical and practical optimization, we present a scalable implementation of a second-order preconditioned method (concretely, a variant of full-matrix Adagr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2002.09018","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2002.09018/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2002.09018","created_at":"2026-07-05T02:20:27.751479+00:00"},{"alias_kind":"arxiv_version","alias_value":"2002.09018v2","created_at":"2026-07-05T02:20:27.751479+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2002.09018","created_at":"2026-07-05T02:20:27.751479+00:00"},{"alias_kind":"pith_short_12","alias_value":"RL4CLPXQ5LWO","created_at":"2026-07-05T02:20:27.751479+00:00"},{"alias_kind":"pith_short_16","alias_value":"RL4CLPXQ5LWOLTG4","created_at":"2026-07-05T02:20:27.751479+00:00"},{"alias_kind":"pith_short_8","alias_value":"RL4CLPXQ","created_at":"2026-07-05T02:20:27.751479+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":24,"internal_anchor_count":3,"sample":[{"citing_arxiv_id":"2607.07204","citing_title":"Restricted Dynamic Geometric Complexity: Path-Space Reduction and M\\\"obius--Jacobi Response","ref_index":3,"is_internal_anchor":true},{"citing_arxiv_id":"2607.07206","citing_title":"Causal Optimizer Interaction Calculus: Hidden Geometric Relaxation and Identifiable Interventions","ref_index":16,"is_internal_anchor":true},{"citing_arxiv_id":"2607.05895","citing_title":"MatrixFSDP: communication-free matrix optimizers under ZeRO-3 parameter sharding","ref_index":2,"is_internal_anchor":true},{"citing_arxiv_id":"2606.12921","citing_title":"LoRA-Muon: Spectral Steepest Descent on the Low-Rank Manifold","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06418","citing_title":"Double Preconditioning (DoPr): Optimization for Test-Time Performance, not Validation Loss","ref_index":286,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04662","citing_title":"Why Muon Outperforms Adam: A Curvature Perspective","ref_index":124,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02365","citing_title":"FOAM: Frequency and Operator Error-Based Adaptive Damping Method for Reducing Staleness-Oriented Error for Shampoo","ref_index":77,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18106","citing_title":"Symmetry-Compatible Principle for Optimizer Design: Embeddings, LM Heads, SwiGLU MLPs, and MoE Routers","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26327","citing_title":"Reparametrizing Shampoo and SOAP for Subspace Basis Updates and BFloat16 Storage","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08352","citing_title":"Convergence Analysis of Newton's Method for Neural Networks in the Overparameterized Limit","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16017","citing_title":"Accelerated Gradient Descent for Faster Convergence with Minimal Overhead","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18106","citing_title":"Symmetry-Compatible Principle for Optimizer Design: Embeddings, LM Heads, SwiGLU MLPs, and MoE Routers","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16184","citing_title":"Runtime-Orchestrated Second-Order Optimization for Scalable LLM Training","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2509.11983","citing_title":"Low-rank Orthogonalization for Large-scale Matrix Optimization with Applications to Foundation Model Training","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2409.20325","citing_title":"Old Optimizer, New Norm: An Anthology","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12994","citing_title":"DP-Muon: Differentially Private Optimization via Matrix-Orthogonalized Momentum","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11316","citing_title":"Error whitening: Why Gauss-Newton outperforms Newton","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08352","citing_title":"Convergence Analysis of Newton's Method for Neural Networks in the Overparameterized Limit","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09176","citing_title":"Navigating LLM Valley: From AdamW to Memory-Efficient and Matrix-Based Optimizers","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09331","citing_title":"Dimension-Free Saddle-Point Escape in Muon","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06316","citing_title":"Pro-KLShampoo: Projected KL-Shampoo with Whitening Recovered by Orthogonalization","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00742","citing_title":"Position: agentic AI orchestration should be Bayes-consistent","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2209.14988","citing_title":"DreamFusion: Text-to-3D using 2D Diffusion","ref_index":89,"is_internal_anchor":false},{"citing_arxiv_id":"2209.14988","citing_title":"DreamFusion: Text-to-3D using 2D Diffusion","ref_index":87,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RL4CLPXQ5LWOLTG4VPEOTYIDUQ","json":"https://pith.science/pith/RL4CLPXQ5LWOLTG4VPEOTYIDUQ.json","graph_json":"https://pith.science/api/pith-number/RL4CLPXQ5LWOLTG4VPEOTYIDUQ/graph.json","events_json":"https://pith.science/api/pith-number/RL4CLPXQ5LWOLTG4VPEOTYIDUQ/events.json","paper":"https://pith.science/paper/RL4CLPXQ"},"agent_actions":{"view_html":"https://pith.science/pith/RL4CLPXQ5LWOLTG4VPEOTYIDUQ","download_json":"https://pith.science/pith/RL4CLPXQ5LWOLTG4VPEOTYIDUQ.json","view_paper":"https://pith.science/paper/RL4CLPXQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2002.09018&json=true","fetch_graph":"https://pith.science/api/pith-number/RL4CLPXQ5LWOLTG4VPEOTYIDUQ/graph.json","fetch_events":"https://pith.science/api/pith-number/RL4CLPXQ5LWOLTG4VPEOTYIDUQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RL4CLPXQ5LWOLTG4VPEOTYIDUQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RL4CLPXQ5LWOLTG4VPEOTYIDUQ/action/storage_attestation","attest_author":"https://pith.science/pith/RL4CLPXQ5LWOLTG4VPEOTYIDUQ/action/author_attestation","sign_citation":"https://pith.science/pith/RL4CLPXQ5LWOLTG4VPEOTYIDUQ/action/citation_signature","submit_replication":"https://pith.science/pith/RL4CLPXQ5LWOLTG4VPEOTYIDUQ/action/replication_record"}},"created_at":"2026-07-05T02:20:27.751479+00:00","updated_at":"2026-07-05T02:20:27.751479+00:00"}