{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:FNMO6C5T3OP3F46UMELFQ3Q7OG","short_pith_number":"pith:FNMO6C5T","schema_version":"1.0","canonical_sha256":"2b58ef0bb3db9fb2f3d46116586e1f71b3440cdd5bdca8af0895374a082add3a","source":{"kind":"arxiv","id":"2407.19804","version":2},"attestation_state":"computed","paper":{"title":"Imputation for prediction: beware of diminishing returns","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"cs.AI","authors_text":"Ga\\\"el Varoquaux (SODA), Marine Le Morvan (SODA)","submitted_at":"2024-07-29T09:01:06Z","abstract_excerpt":"Missing values are prevalent across various fields, posing challenges for training and deploying predictive models. In this context, imputation is a common practice, driven by the hope that accurate imputations will enhance predictions. However, recent theoretical and empirical studies indicate that simple constant imputation can be consistent and competitive. This empirical study aims at clarifying if and when investing in advanced imputation methods yields significantly better predictions. Relating imputation and predictive accuracies across combinations of imputation and predictive models o"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.19804","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2024-07-29T09:01:06Z","cross_cats_sorted":["cs.LG","stat.ML"],"title_canon_sha256":"f517d11ed616180b4880e848aae0ff7ce7ac54660e38257915569b2dbc359f50","abstract_canon_sha256":"209fc9deae70e2aa2328e841b5674732805ffc9a90389f743d533ce27ec9dfad"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:17:09.996811Z","signature_b64":"m6vIzTMRm9Qc3kOCDMCBwhHbc+hMa2tXe8sgG9g5STj/VIc4Ek06zAlqVzHM5Ad1InFA3An3+6X2DaWhPef1Cw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2b58ef0bb3db9fb2f3d46116586e1f71b3440cdd5bdca8af0895374a082add3a","last_reissued_at":"2026-07-05T10:17:09.996311Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:17:09.996311Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Imputation for prediction: beware of diminishing returns","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"cs.AI","authors_text":"Ga\\\"el Varoquaux (SODA), Marine Le Morvan (SODA)","submitted_at":"2024-07-29T09:01:06Z","abstract_excerpt":"Missing values are prevalent across various fields, posing challenges for training and deploying predictive models. In this context, imputation is a common practice, driven by the hope that accurate imputations will enhance predictions. However, recent theoretical and empirical studies indicate that simple constant imputation can be consistent and competitive. This empirical study aims at clarifying if and when investing in advanced imputation methods yields significantly better predictions. Relating imputation and predictive accuracies across combinations of imputation and predictive models o"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.19804","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.19804/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.19804","created_at":"2026-07-05T10:17:09.996374+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.19804v2","created_at":"2026-07-05T10:17:09.996374+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.19804","created_at":"2026-07-05T10:17:09.996374+00:00"},{"alias_kind":"pith_short_12","alias_value":"FNMO6C5T3OP3","created_at":"2026-07-05T10:17:09.996374+00:00"},{"alias_kind":"pith_short_16","alias_value":"FNMO6C5T3OP3F46U","created_at":"2026-07-05T10:17:09.996374+00:00"},{"alias_kind":"pith_short_8","alias_value":"FNMO6C5T","created_at":"2026-07-05T10:17:09.996374+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07767","citing_title":"Distributionally Faithful Imputation via Positive Semi-Definite Kernel Density Estimation","ref_index":13,"is_internal_anchor":true},{"citing_arxiv_id":"2606.02384","citing_title":"TabPrep: Closing the Feature Engineering Gap in Tabular Benchmarks","ref_index":44,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FNMO6C5T3OP3F46UMELFQ3Q7OG","json":"https://pith.science/pith/FNMO6C5T3OP3F46UMELFQ3Q7OG.json","graph_json":"https://pith.science/api/pith-number/FNMO6C5T3OP3F46UMELFQ3Q7OG/graph.json","events_json":"https://pith.science/api/pith-number/FNMO6C5T3OP3F46UMELFQ3Q7OG/events.json","paper":"https://pith.science/paper/FNMO6C5T"},"agent_actions":{"view_html":"https://pith.science/pith/FNMO6C5T3OP3F46UMELFQ3Q7OG","download_json":"https://pith.science/pith/FNMO6C5T3OP3F46UMELFQ3Q7OG.json","view_paper":"https://pith.science/paper/FNMO6C5T","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.19804&json=true","fetch_graph":"https://pith.science/api/pith-number/FNMO6C5T3OP3F46UMELFQ3Q7OG/graph.json","fetch_events":"https://pith.science/api/pith-number/FNMO6C5T3OP3F46UMELFQ3Q7OG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FNMO6C5T3OP3F46UMELFQ3Q7OG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FNMO6C5T3OP3F46UMELFQ3Q7OG/action/storage_attestation","attest_author":"https://pith.science/pith/FNMO6C5T3OP3F46UMELFQ3Q7OG/action/author_attestation","sign_citation":"https://pith.science/pith/FNMO6C5T3OP3F46UMELFQ3Q7OG/action/citation_signature","submit_replication":"https://pith.science/pith/FNMO6C5T3OP3F46UMELFQ3Q7OG/action/replication_record"}},"created_at":"2026-07-05T10:17:09.996374+00:00","updated_at":"2026-07-05T10:17:09.996374+00:00"}