{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:AZH6YPETEMX54JYHCDC7SUEYWK","short_pith_number":"pith:AZH6YPET","schema_version":"1.0","canonical_sha256":"064fec3c93232fde270710c5f95098b2ae21e906c06dc82d46fe11defaac4f8f","source":{"kind":"arxiv","id":"1906.00150","version":5},"attestation_state":"computed","paper":{"title":"Why Not to Use Zero Imputation? Correcting Sparsity Bias in Training Neural Networks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Eunho Yang, Joonyoung Yi, Juhyuk Lee, Kwang Joon Kim, Sung Ju Hwang","submitted_at":"2019-06-01T04:03:53Z","abstract_excerpt":"Handling missing data is one of the most fundamental problems in machine learning. Among many approaches, the simplest and most intuitive way is zero imputation, which treats the value of a missing entry simply as zero. However, many studies have experimentally confirmed that zero imputation results in suboptimal performances in training neural networks. Yet, none of the existing work has explained what brings such performance degradations. In this paper, we introduce the variable sparsity problem (VSP), which describes a phenomenon where the output of a predictive model largely varies with re"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1906.00150","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-06-01T04:03:53Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"c97f67547020fac18f4ac11458aee93a73a950de8d8985b90a1c7be391976575","abstract_canon_sha256":"220cfe2a769d115f5a71a71879d14633db7c8eb62dca3eea2b80bdecfb007f6b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:38:45.285829Z","signature_b64":"woiW5plmGi3PV2jLD6xRx76+CQwKjfjDJxLJEhWxw3aQlxrMHdzM7ctey2f6bh7/5ZDv1AntWpFX0QhKbB7OBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"064fec3c93232fde270710c5f95098b2ae21e906c06dc82d46fe11defaac4f8f","last_reissued_at":"2026-07-05T00:38:45.285357Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:38:45.285357Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Why Not to Use Zero Imputation? Correcting Sparsity Bias in Training Neural Networks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Eunho Yang, Joonyoung Yi, Juhyuk Lee, Kwang Joon Kim, Sung Ju Hwang","submitted_at":"2019-06-01T04:03:53Z","abstract_excerpt":"Handling missing data is one of the most fundamental problems in machine learning. Among many approaches, the simplest and most intuitive way is zero imputation, which treats the value of a missing entry simply as zero. However, many studies have experimentally confirmed that zero imputation results in suboptimal performances in training neural networks. Yet, none of the existing work has explained what brings such performance degradations. In this paper, we introduce the variable sparsity problem (VSP), which describes a phenomenon where the output of a predictive model largely varies with re"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1906.00150","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1906.00150/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1906.00150","created_at":"2026-07-05T00:38:45.285415+00:00"},{"alias_kind":"arxiv_version","alias_value":"1906.00150v5","created_at":"2026-07-05T00:38:45.285415+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1906.00150","created_at":"2026-07-05T00:38:45.285415+00:00"},{"alias_kind":"pith_short_12","alias_value":"AZH6YPETEMX5","created_at":"2026-07-05T00:38:45.285415+00:00"},{"alias_kind":"pith_short_16","alias_value":"AZH6YPETEMX54JYH","created_at":"2026-07-05T00:38:45.285415+00:00"},{"alias_kind":"pith_short_8","alias_value":"AZH6YPET","created_at":"2026-07-05T00:38:45.285415+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03347","citing_title":"AugMask: Training Diffusion Models on Incomplete Tabular Data via Stochastic Augmentation and Masking","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AZH6YPETEMX54JYHCDC7SUEYWK","json":"https://pith.science/pith/AZH6YPETEMX54JYHCDC7SUEYWK.json","graph_json":"https://pith.science/api/pith-number/AZH6YPETEMX54JYHCDC7SUEYWK/graph.json","events_json":"https://pith.science/api/pith-number/AZH6YPETEMX54JYHCDC7SUEYWK/events.json","paper":"https://pith.science/paper/AZH6YPET"},"agent_actions":{"view_html":"https://pith.science/pith/AZH6YPETEMX54JYHCDC7SUEYWK","download_json":"https://pith.science/pith/AZH6YPETEMX54JYHCDC7SUEYWK.json","view_paper":"https://pith.science/paper/AZH6YPET","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1906.00150&json=true","fetch_graph":"https://pith.science/api/pith-number/AZH6YPETEMX54JYHCDC7SUEYWK/graph.json","fetch_events":"https://pith.science/api/pith-number/AZH6YPETEMX54JYHCDC7SUEYWK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AZH6YPETEMX54JYHCDC7SUEYWK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AZH6YPETEMX54JYHCDC7SUEYWK/action/storage_attestation","attest_author":"https://pith.science/pith/AZH6YPETEMX54JYHCDC7SUEYWK/action/author_attestation","sign_citation":"https://pith.science/pith/AZH6YPETEMX54JYHCDC7SUEYWK/action/citation_signature","submit_replication":"https://pith.science/pith/AZH6YPETEMX54JYHCDC7SUEYWK/action/replication_record"}},"created_at":"2026-07-05T00:38:45.285415+00:00","updated_at":"2026-07-05T00:38:45.285415+00:00"}