{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:2NWJYJSZN6K4OTQHGV3PEIWRCK","short_pith_number":"pith:2NWJYJSZ","schema_version":"1.0","canonical_sha256":"d36c9c26596f95c74e073576f222d112bd31d451bbac73dc396bd430cbc3f84f","source":{"kind":"arxiv","id":"2410.21265","version":2},"attestation_state":"computed","paper":{"title":"Modular Duality in Deep Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.NE","stat.ML"],"primary_cat":"cs.LG","authors_text":"Jeremy Bernstein, Laker Newhouse","submitted_at":"2024-10-28T17:57:31Z","abstract_excerpt":"An old idea in optimization theory says that since the gradient is a dual vector it may not be subtracted from the weights without first being mapped to the primal space where the weights reside. We take this idea seriously in this paper and construct such a duality map for general neural networks. Our map, which we call modular dualization, forms a unifying theoretical basis for training algorithms that are a) fast and b) scalable. Modular dualization involves first assigning operator norms to layers based on the semantics of each layer, and then using these layerwise norms to recursively ind"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.21265","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-10-28T17:57:31Z","cross_cats_sorted":["cs.NE","stat.ML"],"title_canon_sha256":"90d86a70b8c5dbd4a394330fd324a900ea4d8767946e96dc1dd18eacf8e39fa7","abstract_canon_sha256":"38bcaf8a7101f2a1a07075ea95159e20aac78cc28fb70ad991c89c4eb321586d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:45:15.852855Z","signature_b64":"thZdB136M1s/p3/wzTOVChsJhwEv19BEgAUCDi8WU8PyPPyKBaIXlE0gXPwGz1aWObPalw+0+1Id2D48T9TUAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d36c9c26596f95c74e073576f222d112bd31d451bbac73dc396bd430cbc3f84f","last_reissued_at":"2026-07-05T09:45:15.852369Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:45:15.852369Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Modular Duality in Deep Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.NE","stat.ML"],"primary_cat":"cs.LG","authors_text":"Jeremy Bernstein, Laker Newhouse","submitted_at":"2024-10-28T17:57:31Z","abstract_excerpt":"An old idea in optimization theory says that since the gradient is a dual vector it may not be subtracted from the weights without first being mapped to the primal space where the weights reside. We take this idea seriously in this paper and construct such a duality map for general neural networks. Our map, which we call modular dualization, forms a unifying theoretical basis for training algorithms that are a) fast and b) scalable. Modular dualization involves first assigning operator norms to layers based on the semantics of each layer, and then using these layerwise norms to recursively ind"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.21265","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.21265/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.21265","created_at":"2026-07-05T09:45:15.852426+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.21265v2","created_at":"2026-07-05T09:45:15.852426+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.21265","created_at":"2026-07-05T09:45:15.852426+00:00"},{"alias_kind":"pith_short_12","alias_value":"2NWJYJSZN6K4","created_at":"2026-07-05T09:45:15.852426+00:00"},{"alias_kind":"pith_short_16","alias_value":"2NWJYJSZN6K4OTQH","created_at":"2026-07-05T09:45:15.852426+00:00"},{"alias_kind":"pith_short_8","alias_value":"2NWJYJSZ","created_at":"2026-07-05T09:45:15.852426+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":18,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25342","citing_title":"Lifelong In-Context Learning with Transformers Requires Parametric Forms of Attention","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01455","citing_title":"Token Geometry","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04662","citing_title":"Why Muon Outperforms Adam: A Curvature Perspective","ref_index":125,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01787","citing_title":"Stochastic convergence of parallel asynchronous adaptive first-order methods","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29176","citing_title":"Dead-Direction Conditioners: Gauge-Equivariant Preconditioning for Deep Networks","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2603.10067","citing_title":"HTMuon: Improving Muon via Heavy-Tailed Spectral Correction","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2502.07529","citing_title":"Training Deep Learning Models with Norm-Constrained LMOs","ref_index":159,"is_internal_anchor":false},{"citing_arxiv_id":"2510.10777","citing_title":"Preconditioned Norms: A Unified Framework for Steepest Descent, Quasi-Newton and Adaptive Methods","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16341","citing_title":"Orth-Dion: Eliminating Geometric Mismatch in Distributed Low-Rank Spectral Optimization","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08949","citing_title":"Muon-OGD: Muon-based Spectral Orthogonal Gradient Projection for LLM Continual Learning","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14200","citing_title":"How to Scale Mixture-of-Experts: From muP to the Maximally Scale-Stable Parameterization","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12492","citing_title":"Pion: A Spectrum-Preserving Optimizer via Orthogonal Equivalence Transformation","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08980","citing_title":"Muon Does Not Converge on Convex Lipschitz Functions","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09238","citing_title":"Intrinsic Muon: Spectral Optimization on Riemannian Matrix Manifolds","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08949","citing_title":"Muon-OGD: Muon-based Spectral Orthogonal Gradient Projection for LLM Continual Learning","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06316","citing_title":"Pro-KLShampoo: Projected KL-Shampoo with Whitening Recovered by Orthogonalization","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07815","citing_title":"OrScale: Orthogonalised Optimization with Layer-Wise Trust-Ratio Scaling","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17423","citing_title":"A unified convergence theory for adaptive first-order methods in the nonconvex case, including AdaNorm, full and diagonal AdaGrad, Shampoo and Muo","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2NWJYJSZN6K4OTQHGV3PEIWRCK","json":"https://pith.science/pith/2NWJYJSZN6K4OTQHGV3PEIWRCK.json","graph_json":"https://pith.science/api/pith-number/2NWJYJSZN6K4OTQHGV3PEIWRCK/graph.json","events_json":"https://pith.science/api/pith-number/2NWJYJSZN6K4OTQHGV3PEIWRCK/events.json","paper":"https://pith.science/paper/2NWJYJSZ"},"agent_actions":{"view_html":"https://pith.science/pith/2NWJYJSZN6K4OTQHGV3PEIWRCK","download_json":"https://pith.science/pith/2NWJYJSZN6K4OTQHGV3PEIWRCK.json","view_paper":"https://pith.science/paper/2NWJYJSZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.21265&json=true","fetch_graph":"https://pith.science/api/pith-number/2NWJYJSZN6K4OTQHGV3PEIWRCK/graph.json","fetch_events":"https://pith.science/api/pith-number/2NWJYJSZN6K4OTQHGV3PEIWRCK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2NWJYJSZN6K4OTQHGV3PEIWRCK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2NWJYJSZN6K4OTQHGV3PEIWRCK/action/storage_attestation","attest_author":"https://pith.science/pith/2NWJYJSZN6K4OTQHGV3PEIWRCK/action/author_attestation","sign_citation":"https://pith.science/pith/2NWJYJSZN6K4OTQHGV3PEIWRCK/action/citation_signature","submit_replication":"https://pith.science/pith/2NWJYJSZN6K4OTQHGV3PEIWRCK/action/replication_record"}},"created_at":"2026-07-05T09:45:15.852426+00:00","updated_at":"2026-07-05T09:45:15.852426+00:00"}