{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:LSCFJASNBNEX5FWNMCHTTENJBF","short_pith_number":"pith:LSCFJASN","schema_version":"1.0","canonical_sha256":"5c8454824d0b497e96cd608f3991a9094b709c2d7bf8d31daf953fa75bc6b0de","source":{"kind":"arxiv","id":"2102.08098","version":3},"attestation_state":"computed","paper":{"title":"GradInit: Learning to Initialize Neural Networks for Stable and Efficient Training","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.CV"],"primary_cat":"cs.LG","authors_text":"Chen Zhu, Kezhi Kong, Renkun Ni, Tom Goldstein, W. Ronny Huang, Zheng Xu","submitted_at":"2021-02-16T11:45:35Z","abstract_excerpt":"Innovations in neural architectures have fostered significant breakthroughs in language modeling and computer vision. Unfortunately, novel architectures often result in challenging hyper-parameter choices and training instability if the network parameters are not properly initialized. A number of architecture-specific initialization schemes have been proposed, but these schemes are not always portable to new architectures. This paper presents GradInit, an automated and architecture agnostic method for initializing neural networks. GradInit is based on a simple heuristic; the norm of each netwo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2102.08098","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-02-16T11:45:35Z","cross_cats_sorted":["cs.CL","cs.CV"],"title_canon_sha256":"841ab423f805cb79030cf215a55b6117659eaa44a199a2e66f1b80169587d1f2","abstract_canon_sha256":"fb77f79ea1172530993ecde2509e9ceddfee105e827b813540811ca14e4e15cd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:34:42.963103Z","signature_b64":"Z2evzvM47RDFehluweFXCronGEarWffs9bIQGp8TcZPz/na+h/ailKemzAL3WpfFVldbi4Q/ObwLtvgrgporAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5c8454824d0b497e96cd608f3991a9094b709c2d7bf8d31daf953fa75bc6b0de","last_reissued_at":"2026-07-05T03:34:42.962632Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:34:42.962632Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GradInit: Learning to Initialize Neural Networks for Stable and Efficient Training","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.CV"],"primary_cat":"cs.LG","authors_text":"Chen Zhu, Kezhi Kong, Renkun Ni, Tom Goldstein, W. Ronny Huang, Zheng Xu","submitted_at":"2021-02-16T11:45:35Z","abstract_excerpt":"Innovations in neural architectures have fostered significant breakthroughs in language modeling and computer vision. Unfortunately, novel architectures often result in challenging hyper-parameter choices and training instability if the network parameters are not properly initialized. A number of architecture-specific initialization schemes have been proposed, but these schemes are not always portable to new architectures. This paper presents GradInit, an automated and architecture agnostic method for initializing neural networks. GradInit is based on a simple heuristic; the norm of each netwo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2102.08098","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2102.08098/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2102.08098","created_at":"2026-07-05T03:34:42.962688+00:00"},{"alias_kind":"arxiv_version","alias_value":"2102.08098v3","created_at":"2026-07-05T03:34:42.962688+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2102.08098","created_at":"2026-07-05T03:34:42.962688+00:00"},{"alias_kind":"pith_short_12","alias_value":"LSCFJASNBNEX","created_at":"2026-07-05T03:34:42.962688+00:00"},{"alias_kind":"pith_short_16","alias_value":"LSCFJASNBNEX5FWN","created_at":"2026-07-05T03:34:42.962688+00:00"},{"alias_kind":"pith_short_8","alias_value":"LSCFJASN","created_at":"2026-07-05T03:34:42.962688+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LSCFJASNBNEX5FWNMCHTTENJBF","json":"https://pith.science/pith/LSCFJASNBNEX5FWNMCHTTENJBF.json","graph_json":"https://pith.science/api/pith-number/LSCFJASNBNEX5FWNMCHTTENJBF/graph.json","events_json":"https://pith.science/api/pith-number/LSCFJASNBNEX5FWNMCHTTENJBF/events.json","paper":"https://pith.science/paper/LSCFJASN"},"agent_actions":{"view_html":"https://pith.science/pith/LSCFJASNBNEX5FWNMCHTTENJBF","download_json":"https://pith.science/pith/LSCFJASNBNEX5FWNMCHTTENJBF.json","view_paper":"https://pith.science/paper/LSCFJASN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2102.08098&json=true","fetch_graph":"https://pith.science/api/pith-number/LSCFJASNBNEX5FWNMCHTTENJBF/graph.json","fetch_events":"https://pith.science/api/pith-number/LSCFJASNBNEX5FWNMCHTTENJBF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LSCFJASNBNEX5FWNMCHTTENJBF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LSCFJASNBNEX5FWNMCHTTENJBF/action/storage_attestation","attest_author":"https://pith.science/pith/LSCFJASNBNEX5FWNMCHTTENJBF/action/author_attestation","sign_citation":"https://pith.science/pith/LSCFJASNBNEX5FWNMCHTTENJBF/action/citation_signature","submit_replication":"https://pith.science/pith/LSCFJASNBNEX5FWNMCHTTENJBF/action/replication_record"}},"created_at":"2026-07-05T03:34:42.962688+00:00","updated_at":"2026-07-05T03:34:42.962688+00:00"}