{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2015:YQ3QBOSGNP5KLOFOM3RS4RMDRT","short_pith_number":"pith:YQ3QBOSG","schema_version":"1.0","canonical_sha256":"c43700ba466bfaa5b8ae66e32e45838cfc6ced2984bf748851f9b42d06986851","source":{"kind":"arxiv","id":"1503.05671","version":7},"attestation_state":"computed","paper":{"title":"Optimizing Neural Networks with Kronecker-factored Approximate Curvature","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.NE","stat.ML"],"primary_cat":"cs.LG","authors_text":"James Martens, Roger Grosse","submitted_at":"2015-03-19T08:30:24Z","abstract_excerpt":"We propose an efficient method for approximating natural gradient descent in neural networks which we call Kronecker-Factored Approximate Curvature (K-FAC). K-FAC is based on an efficiently invertible approximation of a neural network's Fisher information matrix which is neither diagonal nor low-rank, and in some cases is completely non-sparse. It is derived by approximating various large blocks of the Fisher (corresponding to entire layers) as being the Kronecker product of two much smaller matrices. While only several times more expensive to compute than the plain stochastic gradient, the up"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1503.05671","kind":"arxiv","version":7},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2015-03-19T08:30:24Z","cross_cats_sorted":["cs.NE","stat.ML"],"title_canon_sha256":"c6e2a0004252f2169e8ce6856f14a7a1fb3368c25bc02256cc194df6ee1efb60","abstract_canon_sha256":"8dec0638b8f821e695a8232e00c15bbef2fb63c3d83e33341f0b215654feda30"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:08:20.040770Z","signature_b64":"YsWqsn9mJrWdbXE8jDdRmqc/6KL7lSgbR11xnPKVaSDLT0KX4koirey9FM8A++qSB1sAGjF0wk70IWwDZJ+PAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c43700ba466bfaa5b8ae66e32e45838cfc6ced2984bf748851f9b42d06986851","last_reissued_at":"2026-07-05T01:08:20.040328Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:08:20.040328Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Optimizing Neural Networks with Kronecker-factored Approximate Curvature","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.NE","stat.ML"],"primary_cat":"cs.LG","authors_text":"James Martens, Roger Grosse","submitted_at":"2015-03-19T08:30:24Z","abstract_excerpt":"We propose an efficient method for approximating natural gradient descent in neural networks which we call Kronecker-Factored Approximate Curvature (K-FAC). K-FAC is based on an efficiently invertible approximation of a neural network's Fisher information matrix which is neither diagonal nor low-rank, and in some cases is completely non-sparse. It is derived by approximating various large blocks of the Fisher (corresponding to entire layers) as being the Kronecker product of two much smaller matrices. While only several times more expensive to compute than the plain stochastic gradient, the up"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1503.05671","kind":"arxiv","version":7},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1503.05671/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1503.05671","created_at":"2026-07-05T01:08:20.040385+00:00"},{"alias_kind":"arxiv_version","alias_value":"1503.05671v7","created_at":"2026-07-05T01:08:20.040385+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1503.05671","created_at":"2026-07-05T01:08:20.040385+00:00"},{"alias_kind":"pith_short_12","alias_value":"YQ3QBOSGNP5K","created_at":"2026-07-05T01:08:20.040385+00:00"},{"alias_kind":"pith_short_16","alias_value":"YQ3QBOSGNP5KLOFO","created_at":"2026-07-05T01:08:20.040385+00:00"},{"alias_kind":"pith_short_8","alias_value":"YQ3QBOSG","created_at":"2026-07-05T01:08:20.040385+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":17,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07494","citing_title":"GIFT: Geometry-Informed Low-precision Gradient Communication for LLM Pretraining","ref_index":13,"is_internal_anchor":true},{"citing_arxiv_id":"2606.21158","citing_title":"Dead-Direction Signatures: A Cheap Spectral Reading of Singular Complexity","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19491","citing_title":"Algebraic Dead Directions in LayerNorm Transformers: A Forward-Pass-Only Diagnostic at LLM Scale","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02194","citing_title":"An Optimisation Framework for the Well-Conditioned Training of Physics-Informed Neural Networks","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00603","citing_title":"Measuring Dead Directions: Decomposing and Classifying Singular Structure off Canonical Alignment","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05957","citing_title":"Dead Directions: Geometric Singular Learning","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30813","citing_title":"Gradient Smoothing: Coupling Layer-wise Updates for Improved Optimization","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27292","citing_title":"Detectability in Diversity: Improved Canary Crafting for Privacy Auditing in One Run","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29304","citing_title":"On subspace-constrained preconditioning for randomized iterative methods","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23591","citing_title":"Quantifying the Agreement Between Data-Influence and Data-Similarity to Understand LLM Behavior","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16184","citing_title":"Runtime-Orchestrated Second-Order Optimization for Scalable LLM Training","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15899","citing_title":"Solving Classical and Quantum Spin Glasses with Deep Boltzmann Quantum States","ref_index":133,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09263","citing_title":"Natural Riemannian gradient for learning functional tensor networks","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05627","citing_title":"Loss-aware state space geometry for quantum variational algorithms","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15554","citing_title":"Natural gradient descent with momentum","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21100","citing_title":"Preconditioned DeltaNet: Curvature-aware Sequence Modeling for Linear Recurrences","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02829","citing_title":"Compress Then Adapt? No, Do It Together via Task-aware Union of Subspaces","ref_index":30,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YQ3QBOSGNP5KLOFOM3RS4RMDRT","json":"https://pith.science/pith/YQ3QBOSGNP5KLOFOM3RS4RMDRT.json","graph_json":"https://pith.science/api/pith-number/YQ3QBOSGNP5KLOFOM3RS4RMDRT/graph.json","events_json":"https://pith.science/api/pith-number/YQ3QBOSGNP5KLOFOM3RS4RMDRT/events.json","paper":"https://pith.science/paper/YQ3QBOSG"},"agent_actions":{"view_html":"https://pith.science/pith/YQ3QBOSGNP5KLOFOM3RS4RMDRT","download_json":"https://pith.science/pith/YQ3QBOSGNP5KLOFOM3RS4RMDRT.json","view_paper":"https://pith.science/paper/YQ3QBOSG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1503.05671&json=true","fetch_graph":"https://pith.science/api/pith-number/YQ3QBOSGNP5KLOFOM3RS4RMDRT/graph.json","fetch_events":"https://pith.science/api/pith-number/YQ3QBOSGNP5KLOFOM3RS4RMDRT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YQ3QBOSGNP5KLOFOM3RS4RMDRT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YQ3QBOSGNP5KLOFOM3RS4RMDRT/action/storage_attestation","attest_author":"https://pith.science/pith/YQ3QBOSGNP5KLOFOM3RS4RMDRT/action/author_attestation","sign_citation":"https://pith.science/pith/YQ3QBOSGNP5KLOFOM3RS4RMDRT/action/citation_signature","submit_replication":"https://pith.science/pith/YQ3QBOSGNP5KLOFOM3RS4RMDRT/action/replication_record"}},"created_at":"2026-07-05T01:08:20.040385+00:00","updated_at":"2026-07-05T01:08:20.040385+00:00"}