{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:YCVQWJD3TWFJ5BTSSHXNACCJQ4","short_pith_number":"pith:YCVQWJD3","schema_version":"1.0","canonical_sha256":"c0ab0b247b9d8a9e867291eed008498730b91b023e28c14a54dde8c92c6fc80a","source":{"kind":"arxiv","id":"2102.07227","version":2},"attestation_state":"computed","paper":{"title":"Learning by Turning: Neural Architecture Aware Optimisation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.NE","authors_text":"Jeremy Bernstein, Markus Meister, Yang Liu, Yisong Yue","submitted_at":"2021-02-14T19:30:40Z","abstract_excerpt":"Descent methods for deep networks are notoriously capricious: they require careful tuning of step size, momentum and weight decay, and which method will work best on a new benchmark is a priori unclear. To address this problem, this paper conducts a combined study of neural architecture and optimisation, leading to a new optimiser called Nero: the neuronal rotator. Nero trains reliably without momentum or weight decay, works in situations where Adam and SGD fail, and requires little to no learning rate tuning. Also, Nero's memory footprint is ~ square root that of Adam or LAMB. Nero combines t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2102.07227","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.NE","submitted_at":"2021-02-14T19:30:40Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"0400eda567f1b5b577f32685a33575b659bfb0f477fc9a9ab15f4a412b2e76c3","abstract_canon_sha256":"42b2a4af07fde2d2c97c1866b93f739941b1438f031947ea2697797568110fcf"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:15:31.485832Z","signature_b64":"CKcIuTVv9kJiwZG42xubQvvMXc/0rB/6QmCLuVn2u/ZYYJdL5lFstQ6dInppHl36XHK1+wNb8RDxTLQABlU5Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c0ab0b247b9d8a9e867291eed008498730b91b023e28c14a54dde8c92c6fc80a","last_reissued_at":"2026-07-05T03:15:31.485334Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:15:31.485334Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning by Turning: Neural Architecture Aware Optimisation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.NE","authors_text":"Jeremy Bernstein, Markus Meister, Yang Liu, Yisong Yue","submitted_at":"2021-02-14T19:30:40Z","abstract_excerpt":"Descent methods for deep networks are notoriously capricious: they require careful tuning of step size, momentum and weight decay, and which method will work best on a new benchmark is a priori unclear. To address this problem, this paper conducts a combined study of neural architecture and optimisation, leading to a new optimiser called Nero: the neuronal rotator. Nero trains reliably without momentum or weight decay, works in situations where Adam and SGD fail, and requires little to no learning rate tuning. Also, Nero's memory footprint is ~ square root that of Adam or LAMB. Nero combines t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2102.07227","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2102.07227/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2102.07227","created_at":"2026-07-05T03:15:31.485393+00:00"},{"alias_kind":"arxiv_version","alias_value":"2102.07227v2","created_at":"2026-07-05T03:15:31.485393+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2102.07227","created_at":"2026-07-05T03:15:31.485393+00:00"},{"alias_kind":"pith_short_12","alias_value":"YCVQWJD3TWFJ","created_at":"2026-07-05T03:15:31.485393+00:00"},{"alias_kind":"pith_short_16","alias_value":"YCVQWJD3TWFJ5BTS","created_at":"2026-07-05T03:15:31.485393+00:00"},{"alias_kind":"pith_short_8","alias_value":"YCVQWJD3","created_at":"2026-07-05T03:15:31.485393+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25971","citing_title":"Improving Neural Network Training by Decoupling the Magnitude and Direction of Weight Vectors","ref_index":124,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04418","citing_title":"Demystifying Manifold Constraints in LLM Pre-training","ref_index":46,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YCVQWJD3TWFJ5BTSSHXNACCJQ4","json":"https://pith.science/pith/YCVQWJD3TWFJ5BTSSHXNACCJQ4.json","graph_json":"https://pith.science/api/pith-number/YCVQWJD3TWFJ5BTSSHXNACCJQ4/graph.json","events_json":"https://pith.science/api/pith-number/YCVQWJD3TWFJ5BTSSHXNACCJQ4/events.json","paper":"https://pith.science/paper/YCVQWJD3"},"agent_actions":{"view_html":"https://pith.science/pith/YCVQWJD3TWFJ5BTSSHXNACCJQ4","download_json":"https://pith.science/pith/YCVQWJD3TWFJ5BTSSHXNACCJQ4.json","view_paper":"https://pith.science/paper/YCVQWJD3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2102.07227&json=true","fetch_graph":"https://pith.science/api/pith-number/YCVQWJD3TWFJ5BTSSHXNACCJQ4/graph.json","fetch_events":"https://pith.science/api/pith-number/YCVQWJD3TWFJ5BTSSHXNACCJQ4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YCVQWJD3TWFJ5BTSSHXNACCJQ4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YCVQWJD3TWFJ5BTSSHXNACCJQ4/action/storage_attestation","attest_author":"https://pith.science/pith/YCVQWJD3TWFJ5BTSSHXNACCJQ4/action/author_attestation","sign_citation":"https://pith.science/pith/YCVQWJD3TWFJ5BTSSHXNACCJQ4/action/citation_signature","submit_replication":"https://pith.science/pith/YCVQWJD3TWFJ5BTSSHXNACCJQ4/action/replication_record"}},"created_at":"2026-07-05T03:15:31.485393+00:00","updated_at":"2026-07-05T03:15:31.485393+00:00"}