{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:CD2T6G3GOBEUXY7ZCFBQ7S43XD","short_pith_number":"pith:CD2T6G3G","schema_version":"1.0","canonical_sha256":"10f53f1b6670494be3f911430fcb9bb8d2070520f2f43a89df8add2c9cfba9f1","source":{"kind":"arxiv","id":"1905.11675","version":2},"attestation_state":"computed","paper":{"title":"Gram-Gauss-Newton Method: Learning Overparameterized Neural Networks for Regression Problems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Di He, Dong Wang, Jikai Hou, Liwei Wang, Ruiqi Gao, Siyu Chen, Tianle Cai, Zhihua Zhang","submitted_at":"2019-05-28T08:30:24Z","abstract_excerpt":"First-order methods such as stochastic gradient descent (SGD) are currently the standard algorithm for training deep neural networks. Second-order methods, despite their better convergence rate, are rarely used in practice due to the prohibitive computational cost in calculating the second-order information. In this paper, we propose a novel Gram-Gauss-Newton (GGN) algorithm to train deep neural networks for regression problems with square loss. Our method draws inspiration from the connection between neural network optimization and kernel regression of neural tangent kernel (NTK). Different f"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1905.11675","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-05-28T08:30:24Z","cross_cats_sorted":["math.OC","stat.ML"],"title_canon_sha256":"684335238dff174f478e03702bd1279f1c1349a1466f33910a7b5181dea53cb2","abstract_canon_sha256":"a0631cc10d80feb1932def726f3295ea40143e93688cb1c7946185905f554a82"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:07:17.138194Z","signature_b64":"2lrnOBaTzQL93jvuXGhgfNTB8rZP/CIAZNRVzQiLR3swfxeYTgisgZzgRi1p1EuhjfLvwSMqDB3wn40rTNmcDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"10f53f1b6670494be3f911430fcb9bb8d2070520f2f43a89df8add2c9cfba9f1","last_reissued_at":"2026-07-05T00:07:17.137697Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:07:17.137697Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Gram-Gauss-Newton Method: Learning Overparameterized Neural Networks for Regression Problems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Di He, Dong Wang, Jikai Hou, Liwei Wang, Ruiqi Gao, Siyu Chen, Tianle Cai, Zhihua Zhang","submitted_at":"2019-05-28T08:30:24Z","abstract_excerpt":"First-order methods such as stochastic gradient descent (SGD) are currently the standard algorithm for training deep neural networks. Second-order methods, despite their better convergence rate, are rarely used in practice due to the prohibitive computational cost in calculating the second-order information. In this paper, we propose a novel Gram-Gauss-Newton (GGN) algorithm to train deep neural networks for regression problems with square loss. Our method draws inspiration from the connection between neural network optimization and kernel regression of neural tangent kernel (NTK). Different f"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1905.11675","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1905.11675/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1905.11675","created_at":"2026-07-05T00:07:17.137755+00:00"},{"alias_kind":"arxiv_version","alias_value":"1905.11675v2","created_at":"2026-07-05T00:07:17.137755+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1905.11675","created_at":"2026-07-05T00:07:17.137755+00:00"},{"alias_kind":"pith_short_12","alias_value":"CD2T6G3GOBEU","created_at":"2026-07-05T00:07:17.137755+00:00"},{"alias_kind":"pith_short_16","alias_value":"CD2T6G3GOBEUXY7Z","created_at":"2026-07-05T00:07:17.137755+00:00"},{"alias_kind":"pith_short_8","alias_value":"CD2T6G3G","created_at":"2026-07-05T00:07:17.137755+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.27779","citing_title":"Global Convergence and Error Propagation in Neural Gradient Flows: A Riemannian Optimization Framework","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08352","citing_title":"Convergence Analysis of Newton's Method for Neural Networks in the Overparameterized Limit","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18180","citing_title":"Canonical Regularisation of Wide Feature-Learning Neural Networks","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2601.03162","citing_title":"On the Convergence Behavior of Preconditioned Gradient Descent Toward the Rich Learning Regime","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08352","citing_title":"Convergence Analysis of Newton's Method for Neural Networks in the Overparameterized Limit","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CD2T6G3GOBEUXY7ZCFBQ7S43XD","json":"https://pith.science/pith/CD2T6G3GOBEUXY7ZCFBQ7S43XD.json","graph_json":"https://pith.science/api/pith-number/CD2T6G3GOBEUXY7ZCFBQ7S43XD/graph.json","events_json":"https://pith.science/api/pith-number/CD2T6G3GOBEUXY7ZCFBQ7S43XD/events.json","paper":"https://pith.science/paper/CD2T6G3G"},"agent_actions":{"view_html":"https://pith.science/pith/CD2T6G3GOBEUXY7ZCFBQ7S43XD","download_json":"https://pith.science/pith/CD2T6G3GOBEUXY7ZCFBQ7S43XD.json","view_paper":"https://pith.science/paper/CD2T6G3G","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1905.11675&json=true","fetch_graph":"https://pith.science/api/pith-number/CD2T6G3GOBEUXY7ZCFBQ7S43XD/graph.json","fetch_events":"https://pith.science/api/pith-number/CD2T6G3GOBEUXY7ZCFBQ7S43XD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CD2T6G3GOBEUXY7ZCFBQ7S43XD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CD2T6G3GOBEUXY7ZCFBQ7S43XD/action/storage_attestation","attest_author":"https://pith.science/pith/CD2T6G3GOBEUXY7ZCFBQ7S43XD/action/author_attestation","sign_citation":"https://pith.science/pith/CD2T6G3GOBEUXY7ZCFBQ7S43XD/action/citation_signature","submit_replication":"https://pith.science/pith/CD2T6G3GOBEUXY7ZCFBQ7S43XD/action/replication_record"}},"created_at":"2026-07-05T00:07:17.137755+00:00","updated_at":"2026-07-05T00:07:17.137755+00:00"}