{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:ASOCMEWYNAUU3TZRNUR3N7W7D3","short_pith_number":"pith:ASOCMEWY","schema_version":"1.0","canonical_sha256":"049c2612d868294dcf316d23b6fedf1ec92f01482596cae8a3a3372caea08570","source":{"kind":"arxiv","id":"2202.07052","version":1},"attestation_state":"computed","paper":{"title":"Orthogonalising gradients to speed up neural network optimisation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Adam Pr\\\"ugel-Bennett, Jonathan Hare, Mark Tuddenham","submitted_at":"2022-02-14T21:46:07Z","abstract_excerpt":"The optimisation of neural networks can be sped up by orthogonalising the gradients before the optimisation step, ensuring the diversification of the learned representations. We orthogonalise the gradients of the layer's components/filters with respect to each other to separate out the intermediate representations. Our method of orthogonalisation allows the weights to be used more flexibly, in contrast to restricting the weights to an orthogonalised sub-space. We tested this method on ImageNet and CIFAR-10 resulting in a large decrease in learning time, and also obtain a speed-up on the semi-s"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2202.07052","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-02-14T21:46:07Z","cross_cats_sorted":[],"title_canon_sha256":"11529140dea5717541b78e556174edb6a322b17f2651c8970a2e15d239a09a7a","abstract_canon_sha256":"9906d70cfd9a82f1bf968202be2ca05817d46c117f9b9903fec90377fdb8e2e0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:56:55.998695Z","signature_b64":"UqAwHLJ1fJmtWIU73eYmh6WY5cNmFQhZd2+/7a/p4j1uqcAuwWPYqYBqALIBZV0jX4gYIC1Phcd4phzYaT0zCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"049c2612d868294dcf316d23b6fedf1ec92f01482596cae8a3a3372caea08570","last_reissued_at":"2026-07-05T03:56:55.998155Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:56:55.998155Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Orthogonalising gradients to speed up neural network optimisation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Adam Pr\\\"ugel-Bennett, Jonathan Hare, Mark Tuddenham","submitted_at":"2022-02-14T21:46:07Z","abstract_excerpt":"The optimisation of neural networks can be sped up by orthogonalising the gradients before the optimisation step, ensuring the diversification of the learned representations. We orthogonalise the gradients of the layer's components/filters with respect to each other to separate out the intermediate representations. Our method of orthogonalisation allows the weights to be used more flexibly, in contrast to restricting the weights to an orthogonalised sub-space. We tested this method on ImageNet and CIFAR-10 resulting in a large decrease in learning time, and also obtain a speed-up on the semi-s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2202.07052","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2202.07052/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2202.07052","created_at":"2026-07-05T03:56:55.998242+00:00"},{"alias_kind":"arxiv_version","alias_value":"2202.07052v1","created_at":"2026-07-05T03:56:55.998242+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2202.07052","created_at":"2026-07-05T03:56:55.998242+00:00"},{"alias_kind":"pith_short_12","alias_value":"ASOCMEWYNAUU","created_at":"2026-07-05T03:56:55.998242+00:00"},{"alias_kind":"pith_short_16","alias_value":"ASOCMEWYNAUU3TZR","created_at":"2026-07-05T03:56:55.998242+00:00"},{"alias_kind":"pith_short_8","alias_value":"ASOCMEWY","created_at":"2026-07-05T03:56:55.998242+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03899","citing_title":"Denoise First, Orthogonalize Later: Understanding Momentum in Muon via Spectral Filtering","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18106","citing_title":"Symmetry-Compatible Principle for Optimizer Design: Embeddings, LM Heads, SwiGLU MLPs, and MoE Routers","ref_index":150,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23871","citing_title":"Move on Muon : A Hamiltonian probability gradient flow perspective of Muon optimizer","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18106","citing_title":"Symmetry-Compatible Principle for Optimizer Design: Embeddings, LM Heads, SwiGLU MLPs, and MoE Routers","ref_index":148,"is_internal_anchor":false},{"citing_arxiv_id":"2509.11983","citing_title":"Low-rank Orthogonalization for Large-scale Matrix Optimization with Applications to Foundation Model Training","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12492","citing_title":"Pion: A Spectrum-Preserving Optimizer via Orthogonal Equivalence Transformation","ref_index":77,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ASOCMEWYNAUU3TZRNUR3N7W7D3","json":"https://pith.science/pith/ASOCMEWYNAUU3TZRNUR3N7W7D3.json","graph_json":"https://pith.science/api/pith-number/ASOCMEWYNAUU3TZRNUR3N7W7D3/graph.json","events_json":"https://pith.science/api/pith-number/ASOCMEWYNAUU3TZRNUR3N7W7D3/events.json","paper":"https://pith.science/paper/ASOCMEWY"},"agent_actions":{"view_html":"https://pith.science/pith/ASOCMEWYNAUU3TZRNUR3N7W7D3","download_json":"https://pith.science/pith/ASOCMEWYNAUU3TZRNUR3N7W7D3.json","view_paper":"https://pith.science/paper/ASOCMEWY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2202.07052&json=true","fetch_graph":"https://pith.science/api/pith-number/ASOCMEWYNAUU3TZRNUR3N7W7D3/graph.json","fetch_events":"https://pith.science/api/pith-number/ASOCMEWYNAUU3TZRNUR3N7W7D3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ASOCMEWYNAUU3TZRNUR3N7W7D3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ASOCMEWYNAUU3TZRNUR3N7W7D3/action/storage_attestation","attest_author":"https://pith.science/pith/ASOCMEWYNAUU3TZRNUR3N7W7D3/action/author_attestation","sign_citation":"https://pith.science/pith/ASOCMEWYNAUU3TZRNUR3N7W7D3/action/citation_signature","submit_replication":"https://pith.science/pith/ASOCMEWYNAUU3TZRNUR3N7W7D3/action/replication_record"}},"created_at":"2026-07-05T03:56:55.998242+00:00","updated_at":"2026-07-05T03:56:55.998242+00:00"}