{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:VCXEXNUL6LTDQSB7CBY3XC2RNJ","short_pith_number":"pith:VCXEXNUL","schema_version":"1.0","canonical_sha256":"a8ae4bb68bf2e638483f1071bb8b516a597dda3d2576120ecb8b69792fd57a9d","source":{"kind":"arxiv","id":"1903.06733","version":3},"attestation_state":"computed","paper":{"title":"Dying ReLU and Initialization: Theory and Numerical Examples","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.LG","math.PR"],"primary_cat":"stat.ML","authors_text":"George Em Karniadakis, Lu Lu, Yanhui Su, Yeonjong Shin","submitted_at":"2019-03-15T18:23:55Z","abstract_excerpt":"The dying ReLU refers to the problem when ReLU neurons become inactive and only output 0 for any input. There are many empirical and heuristic explanations of why ReLU neurons die. However, little is known about its theoretical analysis. In this paper, we rigorously prove that a deep ReLU network will eventually die in probability as the depth goes to infinite. Several methods have been proposed to alleviate the dying ReLU. Perhaps, one of the simplest treatments is to modify the initialization procedure. One common way of initializing weights and biases uses symmetric probability distribution"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1903.06733","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"stat.ML","submitted_at":"2019-03-15T18:23:55Z","cross_cats_sorted":["cs.LG","math.PR"],"title_canon_sha256":"8b8773267b9334380c1c5bc738714f36c70185462830422173e34b0645e8620b","abstract_canon_sha256":"c692092728c5f377b6ac633d597af76baed46cc003b0f76160ea61634a389259"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:02:30.028489Z","signature_b64":"z0QYIhI2uyvFgfM8i9ENcbLbAPMV8tj03QNCmIKPoMBfbBiIxZQl0PkMm8sFbY087+Fm43Gd2n0gFicImtfaCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a8ae4bb68bf2e638483f1071bb8b516a597dda3d2576120ecb8b69792fd57a9d","last_reissued_at":"2026-07-05T02:02:30.028018Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:02:30.028018Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Dying ReLU and Initialization: Theory and Numerical Examples","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.LG","math.PR"],"primary_cat":"stat.ML","authors_text":"George Em Karniadakis, Lu Lu, Yanhui Su, Yeonjong Shin","submitted_at":"2019-03-15T18:23:55Z","abstract_excerpt":"The dying ReLU refers to the problem when ReLU neurons become inactive and only output 0 for any input. There are many empirical and heuristic explanations of why ReLU neurons die. However, little is known about its theoretical analysis. In this paper, we rigorously prove that a deep ReLU network will eventually die in probability as the depth goes to infinite. Several methods have been proposed to alleviate the dying ReLU. Perhaps, one of the simplest treatments is to modify the initialization procedure. One common way of initializing weights and biases uses symmetric probability distribution"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1903.06733","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1903.06733/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1903.06733","created_at":"2026-07-05T02:02:30.028086+00:00"},{"alias_kind":"arxiv_version","alias_value":"1903.06733v3","created_at":"2026-07-05T02:02:30.028086+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1903.06733","created_at":"2026-07-05T02:02:30.028086+00:00"},{"alias_kind":"pith_short_12","alias_value":"VCXEXNUL6LTD","created_at":"2026-07-05T02:02:30.028086+00:00"},{"alias_kind":"pith_short_16","alias_value":"VCXEXNUL6LTDQSB7","created_at":"2026-07-05T02:02:30.028086+00:00"},{"alias_kind":"pith_short_8","alias_value":"VCXEXNUL","created_at":"2026-07-05T02:02:30.028086+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23729","citing_title":"Interpretable Material Spatial Intelligence for Discovery of Governing Microstructural Features","ref_index":96,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21297","citing_title":"NASDAQ: Normalized Observation Space Dynamics-Augmented Q-Learning","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09762","citing_title":"Preserving Plasticity in Continual Learning via Dynamical Isometry","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17659","citing_title":"Bug or Feature$^2$: Weight Drift, Activation Sparsity and Spikes","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17659","citing_title":"Bug or Feature$^2$: Weight Drift, Activation Sparsity and Spikes","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"1910.03193","citing_title":"DeepONet: Learning nonlinear operators for identifying differential equations based on the universal approximation theorem of operators","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VCXEXNUL6LTDQSB7CBY3XC2RNJ","json":"https://pith.science/pith/VCXEXNUL6LTDQSB7CBY3XC2RNJ.json","graph_json":"https://pith.science/api/pith-number/VCXEXNUL6LTDQSB7CBY3XC2RNJ/graph.json","events_json":"https://pith.science/api/pith-number/VCXEXNUL6LTDQSB7CBY3XC2RNJ/events.json","paper":"https://pith.science/paper/VCXEXNUL"},"agent_actions":{"view_html":"https://pith.science/pith/VCXEXNUL6LTDQSB7CBY3XC2RNJ","download_json":"https://pith.science/pith/VCXEXNUL6LTDQSB7CBY3XC2RNJ.json","view_paper":"https://pith.science/paper/VCXEXNUL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1903.06733&json=true","fetch_graph":"https://pith.science/api/pith-number/VCXEXNUL6LTDQSB7CBY3XC2RNJ/graph.json","fetch_events":"https://pith.science/api/pith-number/VCXEXNUL6LTDQSB7CBY3XC2RNJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VCXEXNUL6LTDQSB7CBY3XC2RNJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VCXEXNUL6LTDQSB7CBY3XC2RNJ/action/storage_attestation","attest_author":"https://pith.science/pith/VCXEXNUL6LTDQSB7CBY3XC2RNJ/action/author_attestation","sign_citation":"https://pith.science/pith/VCXEXNUL6LTDQSB7CBY3XC2RNJ/action/citation_signature","submit_replication":"https://pith.science/pith/VCXEXNUL6LTDQSB7CBY3XC2RNJ/action/replication_record"}},"created_at":"2026-07-05T02:02:30.028086+00:00","updated_at":"2026-07-05T02:02:30.028086+00:00"}