{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:YNQW7LCH5GA7A2YUZ6QTGECFQJ","short_pith_number":"pith:YNQW7LCH","schema_version":"1.0","canonical_sha256":"c3616fac47e981f06b14cfa13310458267bb7883a6d687255c9907d3c65ae83f","source":{"kind":"arxiv","id":"2403.14687","version":1},"attestation_state":"computed","paper":{"title":"On the Performance of Imputation Techniques for Missing Values on Healthcare Datasets","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Babu Sena Paul, Luke Oluwaseye Joel, Wesley Doorsamy","submitted_at":"2024-03-13T18:07:17Z","abstract_excerpt":"Missing values or data is one popular characteristic of real-world datasets, especially healthcare data. This could be frustrating when using machine learning algorithms on such datasets, simply because most machine learning models perform poorly in the presence of missing values. The aim of this study is to compare the performance of seven imputation techniques, namely Mean imputation, Median Imputation, Last Observation carried Forward (LOCF) imputation, K-Nearest Neighbor (KNN) imputation, Interpolation imputation, Missforest imputation, and Multiple imputation by Chained Equations (MICE), "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.14687","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2024-03-13T18:07:17Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"20048b368de3f34a26a91c2d7daa7c2f7394a0025a05a9e3adb2f1ac23c3650b","abstract_canon_sha256":"fbede491c69c6541cb67afb4217f5e31a499dc9cc2905a019ef652cc1d8d4c68"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:59:24.579064Z","signature_b64":"6eP+KKVGw4xbejbOHTccC0NCrDgmVQap+k4aA5BIy4d4lAX6PjzgUJmMqBvwu94ZIkwcqGEKlFOODGmlkfUgBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c3616fac47e981f06b14cfa13310458267bb7883a6d687255c9907d3c65ae83f","last_reissued_at":"2026-07-05T07:59:24.578581Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:59:24.578581Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On the Performance of Imputation Techniques for Missing Values on Healthcare Datasets","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Babu Sena Paul, Luke Oluwaseye Joel, Wesley Doorsamy","submitted_at":"2024-03-13T18:07:17Z","abstract_excerpt":"Missing values or data is one popular characteristic of real-world datasets, especially healthcare data. This could be frustrating when using machine learning algorithms on such datasets, simply because most machine learning models perform poorly in the presence of missing values. The aim of this study is to compare the performance of seven imputation techniques, namely Mean imputation, Median Imputation, Last Observation carried Forward (LOCF) imputation, K-Nearest Neighbor (KNN) imputation, Interpolation imputation, Missforest imputation, and Multiple imputation by Chained Equations (MICE), "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.14687","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.14687/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.14687","created_at":"2026-07-05T07:59:24.578640+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.14687v1","created_at":"2026-07-05T07:59:24.578640+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.14687","created_at":"2026-07-05T07:59:24.578640+00:00"},{"alias_kind":"pith_short_12","alias_value":"YNQW7LCH5GA7","created_at":"2026-07-05T07:59:24.578640+00:00"},{"alias_kind":"pith_short_16","alias_value":"YNQW7LCH5GA7A2YU","created_at":"2026-07-05T07:59:24.578640+00:00"},{"alias_kind":"pith_short_8","alias_value":"YNQW7LCH","created_at":"2026-07-05T07:59:24.578640+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2502.14270","citing_title":"Predicting Fetal Birthweight from High Dimensional Data using Advanced Machine Learning","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YNQW7LCH5GA7A2YUZ6QTGECFQJ","json":"https://pith.science/pith/YNQW7LCH5GA7A2YUZ6QTGECFQJ.json","graph_json":"https://pith.science/api/pith-number/YNQW7LCH5GA7A2YUZ6QTGECFQJ/graph.json","events_json":"https://pith.science/api/pith-number/YNQW7LCH5GA7A2YUZ6QTGECFQJ/events.json","paper":"https://pith.science/paper/YNQW7LCH"},"agent_actions":{"view_html":"https://pith.science/pith/YNQW7LCH5GA7A2YUZ6QTGECFQJ","download_json":"https://pith.science/pith/YNQW7LCH5GA7A2YUZ6QTGECFQJ.json","view_paper":"https://pith.science/paper/YNQW7LCH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.14687&json=true","fetch_graph":"https://pith.science/api/pith-number/YNQW7LCH5GA7A2YUZ6QTGECFQJ/graph.json","fetch_events":"https://pith.science/api/pith-number/YNQW7LCH5GA7A2YUZ6QTGECFQJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YNQW7LCH5GA7A2YUZ6QTGECFQJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YNQW7LCH5GA7A2YUZ6QTGECFQJ/action/storage_attestation","attest_author":"https://pith.science/pith/YNQW7LCH5GA7A2YUZ6QTGECFQJ/action/author_attestation","sign_citation":"https://pith.science/pith/YNQW7LCH5GA7A2YUZ6QTGECFQJ/action/citation_signature","submit_replication":"https://pith.science/pith/YNQW7LCH5GA7A2YUZ6QTGECFQJ/action/replication_record"}},"created_at":"2026-07-05T07:59:24.578640+00:00","updated_at":"2026-07-05T07:59:24.578640+00:00"}