{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:H2AE2BUGZ5JF2SNDUV2F6ICOLK","short_pith_number":"pith:H2AE2BUG","schema_version":"1.0","canonical_sha256":"3e804d0686cf525d49a3a5745f204e5aa52e1772409d94801bd5cfbc2b8ab09b","source":{"kind":"arxiv","id":"2507.15681","version":1},"attestation_state":"computed","paper":{"title":"Missing value imputation with adversarial random forests -- MissARF","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"stat.ML","authors_text":"David S. Watson, Jan Kapar, Marvin N. Wright, Pegah Golchian","submitted_at":"2025-07-21T14:44:51Z","abstract_excerpt":"Handling missing values is a common challenge in biostatistical analyses, typically addressed by imputation methods. We propose a novel, fast, and easy-to-use imputation method called missing value imputation with adversarial random forests (MissARF), based on generative machine learning, that provides both single and multiple imputation. MissARF employs adversarial random forest (ARF) for density estimation and data synthesis. To impute a missing value of an observation, we condition on the non-missing values and sample from the estimated conditional distribution generated by ARF. Our experim"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.15681","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"stat.ML","submitted_at":"2025-07-21T14:44:51Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"9e5f0f2aa64f5a7fb3f226ef6f172b0972067f999be389f1ae8a6aa5791c0ad6","abstract_canon_sha256":"e9dfe5c9f8d34bc9c02494d75749ce530d63b29cc9b275800d18c7c7598a84ac"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:40:41.913092Z","signature_b64":"TQiltsA+piiKfPcv/L1sdgPmzAPlAjNTr4PchnpOmrQ84YQsyuQBzjTiKhUC3GauVuDrW29q12W2w+W7jul4Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3e804d0686cf525d49a3a5745f204e5aa52e1772409d94801bd5cfbc2b8ab09b","last_reissued_at":"2026-07-05T11:40:41.912609Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:40:41.912609Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Missing value imputation with adversarial random forests -- MissARF","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"stat.ML","authors_text":"David S. Watson, Jan Kapar, Marvin N. Wright, Pegah Golchian","submitted_at":"2025-07-21T14:44:51Z","abstract_excerpt":"Handling missing values is a common challenge in biostatistical analyses, typically addressed by imputation methods. We propose a novel, fast, and easy-to-use imputation method called missing value imputation with adversarial random forests (MissARF), based on generative machine learning, that provides both single and multiple imputation. MissARF employs adversarial random forest (ARF) for density estimation and data synthesis. To impute a missing value of an observation, we condition on the non-missing values and sample from the estimated conditional distribution generated by ARF. Our experim"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.15681","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.15681/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.15681","created_at":"2026-07-05T11:40:41.912666+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.15681v1","created_at":"2026-07-05T11:40:41.912666+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.15681","created_at":"2026-07-05T11:40:41.912666+00:00"},{"alias_kind":"pith_short_12","alias_value":"H2AE2BUGZ5JF","created_at":"2026-07-05T11:40:41.912666+00:00"},{"alias_kind":"pith_short_16","alias_value":"H2AE2BUGZ5JF2SND","created_at":"2026-07-05T11:40:41.912666+00:00"},{"alias_kind":"pith_short_8","alias_value":"H2AE2BUG","created_at":"2026-07-05T11:40:41.912666+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2508.14936","citing_title":"Can synthetic data reproduce real-world findings in epidemiology? A replication study using adversarial random forests","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22980","citing_title":"Testing independence in the presence of missing data: high-dimensional case","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/H2AE2BUGZ5JF2SNDUV2F6ICOLK","json":"https://pith.science/pith/H2AE2BUGZ5JF2SNDUV2F6ICOLK.json","graph_json":"https://pith.science/api/pith-number/H2AE2BUGZ5JF2SNDUV2F6ICOLK/graph.json","events_json":"https://pith.science/api/pith-number/H2AE2BUGZ5JF2SNDUV2F6ICOLK/events.json","paper":"https://pith.science/paper/H2AE2BUG"},"agent_actions":{"view_html":"https://pith.science/pith/H2AE2BUGZ5JF2SNDUV2F6ICOLK","download_json":"https://pith.science/pith/H2AE2BUGZ5JF2SNDUV2F6ICOLK.json","view_paper":"https://pith.science/paper/H2AE2BUG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.15681&json=true","fetch_graph":"https://pith.science/api/pith-number/H2AE2BUGZ5JF2SNDUV2F6ICOLK/graph.json","fetch_events":"https://pith.science/api/pith-number/H2AE2BUGZ5JF2SNDUV2F6ICOLK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/H2AE2BUGZ5JF2SNDUV2F6ICOLK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/H2AE2BUGZ5JF2SNDUV2F6ICOLK/action/storage_attestation","attest_author":"https://pith.science/pith/H2AE2BUGZ5JF2SNDUV2F6ICOLK/action/author_attestation","sign_citation":"https://pith.science/pith/H2AE2BUGZ5JF2SNDUV2F6ICOLK/action/citation_signature","submit_replication":"https://pith.science/pith/H2AE2BUGZ5JF2SNDUV2F6ICOLK/action/replication_record"}},"created_at":"2026-07-05T11:40:41.912666+00:00","updated_at":"2026-07-05T11:40:41.912666+00:00"}