{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:XYNUTFIPPLV4P6S3VSUL7SD7ZC","short_pith_number":"pith:XYNUTFIP","schema_version":"1.0","canonical_sha256":"be1b49950f7aebc7fa5baca8bfc87fc89a221467cdee8676147e506aa2cbc6e1","source":{"kind":"arxiv","id":"2006.02409","version":4},"attestation_state":"computed","paper":{"title":"On the Promise of the Stochastic Generalized Gauss-Newton Method for Training DNNs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Andrea Zanelli, Frank Hutter, Matilde Gargiani, Moritz Diehl","submitted_at":"2020-06-03T17:35:54Z","abstract_excerpt":"Following early work on Hessian-free methods for deep learning, we study a stochastic generalized Gauss-Newton method (SGN) for training DNNs. SGN is a second-order optimization method, with efficient iterations, that we demonstrate to often require substantially fewer iterations than standard SGD to converge. As the name suggests, SGN uses a Gauss-Newton approximation for the Hessian matrix, and, in order to compute an approximate search direction, relies on the conjugate gradient method combined with forward and reverse automatic differentiation. Despite the success of SGD and its first-orde"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2006.02409","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-06-03T17:35:54Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"db263e06384b2a14e2975bdf90439a9f8ad4d982be806a21e4fa2ce29935a746","abstract_canon_sha256":"a77e190b117fd8a78b63b1bbf642c9f573c1327eaf293df2abe8e3f1fb3da5a3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:09:07.456390Z","signature_b64":"sO5kuYIFrl32SJZ6806gRLV2WhPOYD7ZnklyKk1eYNXpTtSMUXcSBISByRutL5l2YN2nkKKzs+WNjI3E13UoDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"be1b49950f7aebc7fa5baca8bfc87fc89a221467cdee8676147e506aa2cbc6e1","last_reissued_at":"2026-07-05T01:09:07.455805Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:09:07.455805Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On the Promise of the Stochastic Generalized Gauss-Newton Method for Training DNNs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Andrea Zanelli, Frank Hutter, Matilde Gargiani, Moritz Diehl","submitted_at":"2020-06-03T17:35:54Z","abstract_excerpt":"Following early work on Hessian-free methods for deep learning, we study a stochastic generalized Gauss-Newton method (SGN) for training DNNs. SGN is a second-order optimization method, with efficient iterations, that we demonstrate to often require substantially fewer iterations than standard SGD to converge. As the name suggests, SGN uses a Gauss-Newton approximation for the Hessian matrix, and, in order to compute an approximate search direction, relies on the conjugate gradient method combined with forward and reverse automatic differentiation. Despite the success of SGD and its first-orde"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2006.02409","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2006.02409/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2006.02409","created_at":"2026-07-05T01:09:07.455882+00:00"},{"alias_kind":"arxiv_version","alias_value":"2006.02409v4","created_at":"2026-07-05T01:09:07.455882+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2006.02409","created_at":"2026-07-05T01:09:07.455882+00:00"},{"alias_kind":"pith_short_12","alias_value":"XYNUTFIPPLV4","created_at":"2026-07-05T01:09:07.455882+00:00"},{"alias_kind":"pith_short_16","alias_value":"XYNUTFIPPLV4P6S3","created_at":"2026-07-05T01:09:07.455882+00:00"},{"alias_kind":"pith_short_8","alias_value":"XYNUTFIP","created_at":"2026-07-05T01:09:07.455882+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2601.03162","citing_title":"On the Convergence Behavior of Preconditioned Gradient Descent Toward the Rich Learning Regime","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06081","citing_title":"Fast Gauss-Newton for Multiclass Cross-Entropy","ref_index":11,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XYNUTFIPPLV4P6S3VSUL7SD7ZC","json":"https://pith.science/pith/XYNUTFIPPLV4P6S3VSUL7SD7ZC.json","graph_json":"https://pith.science/api/pith-number/XYNUTFIPPLV4P6S3VSUL7SD7ZC/graph.json","events_json":"https://pith.science/api/pith-number/XYNUTFIPPLV4P6S3VSUL7SD7ZC/events.json","paper":"https://pith.science/paper/XYNUTFIP"},"agent_actions":{"view_html":"https://pith.science/pith/XYNUTFIPPLV4P6S3VSUL7SD7ZC","download_json":"https://pith.science/pith/XYNUTFIPPLV4P6S3VSUL7SD7ZC.json","view_paper":"https://pith.science/paper/XYNUTFIP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2006.02409&json=true","fetch_graph":"https://pith.science/api/pith-number/XYNUTFIPPLV4P6S3VSUL7SD7ZC/graph.json","fetch_events":"https://pith.science/api/pith-number/XYNUTFIPPLV4P6S3VSUL7SD7ZC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XYNUTFIPPLV4P6S3VSUL7SD7ZC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XYNUTFIPPLV4P6S3VSUL7SD7ZC/action/storage_attestation","attest_author":"https://pith.science/pith/XYNUTFIPPLV4P6S3VSUL7SD7ZC/action/author_attestation","sign_citation":"https://pith.science/pith/XYNUTFIPPLV4P6S3VSUL7SD7ZC/action/citation_signature","submit_replication":"https://pith.science/pith/XYNUTFIPPLV4P6S3VSUL7SD7ZC/action/replication_record"}},"created_at":"2026-07-05T01:09:07.455882+00:00","updated_at":"2026-07-05T01:09:07.455882+00:00"}