{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2013:IPSZLZHDWKHHJFTXMVTG2RUE3M","short_pith_number":"pith:IPSZLZHD","schema_version":"1.0","canonical_sha256":"43e595e4e3b28e74967765666d4684db2054514f7fe53f3fb8a55858907025f6","source":{"kind":"arxiv","id":"1301.3584","version":7},"attestation_state":"computed","paper":{"title":"Revisiting Natural Gradient for Deep Networks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.NA"],"primary_cat":"cs.LG","authors_text":"Razvan Pascanu, Yoshua Bengio","submitted_at":"2013-01-16T04:47:02Z","abstract_excerpt":"We evaluate natural gradient, an algorithm originally proposed in Amari (1997), for learning deep models. The contributions of this paper are as follows. We show the connection between natural gradient and three other recently proposed methods for training deep models: Hessian-Free (Martens, 2010), Krylov Subspace Descent (Vinyals and Povey, 2012) and TONGA (Le Roux et al., 2008). We describe how one can use unlabeled data to improve the generalization error obtained by natural gradient and empirically evaluate the robustness of the algorithm to the ordering of the training set compared to sto"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1301.3584","kind":"arxiv","version":7},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2013-01-16T04:47:02Z","cross_cats_sorted":["cs.NA"],"title_canon_sha256":"25709ee60081e4b4893f026bc1c89c75c0a0a9f38314de09ee2ab45599d4a427","abstract_canon_sha256":"f0eccd6f4c3c86edaad6ac332adf3b7b71980a35995824cea6fac71f8bc934c8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-18T02:58:57.462134Z","signature_b64":"HGr8a4XRSSinxfnOcRuV1rpvKH9+8whwOfNDU7s2jyjbhq68TvGFapeB9L+gzgETWAzcOhxxbcRDHAPD2Y5XDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"43e595e4e3b28e74967765666d4684db2054514f7fe53f3fb8a55858907025f6","last_reissued_at":"2026-05-18T02:58:57.461367Z","signature_status":"signed_v1","first_computed_at":"2026-05-18T02:58:57.461367Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Revisiting Natural Gradient for Deep Networks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.NA"],"primary_cat":"cs.LG","authors_text":"Razvan Pascanu, Yoshua Bengio","submitted_at":"2013-01-16T04:47:02Z","abstract_excerpt":"We evaluate natural gradient, an algorithm originally proposed in Amari (1997), for learning deep models. The contributions of this paper are as follows. We show the connection between natural gradient and three other recently proposed methods for training deep models: Hessian-Free (Martens, 2010), Krylov Subspace Descent (Vinyals and Povey, 2012) and TONGA (Le Roux et al., 2008). We describe how one can use unlabeled data to improve the generalization error obtained by natural gradient and empirically evaluate the robustness of the algorithm to the ordering of the training set compared to sto"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1301.3584","kind":"arxiv","version":7},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1301.3584","created_at":"2026-05-18T02:58:57.461496+00:00"},{"alias_kind":"arxiv_version","alias_value":"1301.3584v7","created_at":"2026-05-18T02:58:57.461496+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1301.3584","created_at":"2026-05-18T02:58:57.461496+00:00"},{"alias_kind":"pith_short_12","alias_value":"IPSZLZHDWKHH","created_at":"2026-05-18T12:27:46.883200+00:00"},{"alias_kind":"pith_short_16","alias_value":"IPSZLZHDWKHHJFTX","created_at":"2026-05-18T12:27:46.883200+00:00"},{"alias_kind":"pith_short_8","alias_value":"IPSZLZHD","created_at":"2026-05-18T12:27:46.883200+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":11,"sample":[{"citing_arxiv_id":"2607.07845","citing_title":"Explaining Near-Zero Hessian Eigenvalues Through Approximate Symmetries in Neural Networks","ref_index":62,"is_internal_anchor":true},{"citing_arxiv_id":"2606.06418","citing_title":"Double Preconditioning (DoPr): Optimization for Test-Time Performance, not Validation Loss","ref_index":142,"is_internal_anchor":true},{"citing_arxiv_id":"2606.03382","citing_title":"Local Guidance, Global Impact: Gaussian-Reshaped Trust Region Unlocks Behavior Transitions","ref_index":32,"is_internal_anchor":true},{"citing_arxiv_id":"2606.04880","citing_title":"MAOAM: Unified Object and Material Selection with Vision-Language Models","ref_index":52,"is_internal_anchor":true},{"citing_arxiv_id":"2605.01046","citing_title":"Learning in the Fisher Subspace: A Guided Initialization for LoRA Fine-Tuning","ref_index":3,"is_internal_anchor":true},{"citing_arxiv_id":"2606.01445","citing_title":"Multiparameter Maximum Information States for Coherent Diffraction Measurements","ref_index":41,"is_internal_anchor":true},{"citing_arxiv_id":"2502.02345","citing_title":"Low Rank Based Subspace Inference for the Laplace Approximation of Bayesian Neural Networks","ref_index":19,"is_internal_anchor":true},{"citing_arxiv_id":"2510.04930","citing_title":"Egalitarian Gradient Descent: A Simple Approach to Accelerated Grokking","ref_index":9,"is_internal_anchor":true},{"citing_arxiv_id":"2605.16165","citing_title":"Second-Order Multi-Level Variance Correction for Modality Competition in Multimodal Models","ref_index":17,"is_internal_anchor":true},{"citing_arxiv_id":"2605.15899","citing_title":"Solving Classical and Quantum Spin Glasses with Deep Boltzmann Quantum States","ref_index":74,"is_internal_anchor":true},{"citing_arxiv_id":"2603.22347","citing_title":"Intelligence Inertia: Physical Isomorphism and Applications","ref_index":20,"is_internal_anchor":true},{"citing_arxiv_id":"2605.04230","citing_title":"Layerwise LQR for Geometry-Aware Optimization of Deep Networks","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04115","citing_title":"Learning reveals invisible structure in low-rank RNNs","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"1412.6980","citing_title":"Adam: A Method for Stochastic Optimization","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IPSZLZHDWKHHJFTXMVTG2RUE3M","json":"https://pith.science/pith/IPSZLZHDWKHHJFTXMVTG2RUE3M.json","graph_json":"https://pith.science/api/pith-number/IPSZLZHDWKHHJFTXMVTG2RUE3M/graph.json","events_json":"https://pith.science/api/pith-number/IPSZLZHDWKHHJFTXMVTG2RUE3M/events.json","paper":"https://pith.science/paper/IPSZLZHD"},"agent_actions":{"view_html":"https://pith.science/pith/IPSZLZHDWKHHJFTXMVTG2RUE3M","download_json":"https://pith.science/pith/IPSZLZHDWKHHJFTXMVTG2RUE3M.json","view_paper":"https://pith.science/paper/IPSZLZHD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1301.3584&json=true","fetch_graph":"https://pith.science/api/pith-number/IPSZLZHDWKHHJFTXMVTG2RUE3M/graph.json","fetch_events":"https://pith.science/api/pith-number/IPSZLZHDWKHHJFTXMVTG2RUE3M/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IPSZLZHDWKHHJFTXMVTG2RUE3M/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IPSZLZHDWKHHJFTXMVTG2RUE3M/action/storage_attestation","attest_author":"https://pith.science/pith/IPSZLZHDWKHHJFTXMVTG2RUE3M/action/author_attestation","sign_citation":"https://pith.science/pith/IPSZLZHDWKHHJFTXMVTG2RUE3M/action/citation_signature","submit_replication":"https://pith.science/pith/IPSZLZHDWKHHJFTXMVTG2RUE3M/action/replication_record"}},"created_at":"2026-05-18T02:58:57.461496+00:00","updated_at":"2026-05-18T02:58:57.461496+00:00"}