{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:6G34BHD2WOLPRCAYAPHVE6M2GG","short_pith_number":"pith:6G34BHD2","schema_version":"1.0","canonical_sha256":"f1b7c09c7ab396f8881803cf52799a31b584f06f7278ef7316ffcc82a7eeb0ae","source":{"kind":"arxiv","id":"2311.14220","version":4},"attestation_state":"computed","paper":{"title":"Assumption-Lean and Data-Adaptive Post-Prediction Inference","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"stat.ME","authors_text":"Jiacheng Miao, Jiwei Zhao, Qiongshi Lu, Xinran Miao, Yixuan Wu","submitted_at":"2023-11-23T22:41:30Z","abstract_excerpt":"A primary challenge facing modern scientific research is the limited availability of gold-standard data which can be costly, labor-intensive, or invasive to obtain. With the rapid development of machine learning (ML), scientists can now employ ML algorithms to predict gold-standard outcomes with variables that are easier to obtain. However, these predicted outcomes are often used directly in subsequent statistical analyses, ignoring imprecision and heterogeneity introduced by the prediction procedure. This will likely result in false positive findings and invalid scientific conclusions. In thi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.14220","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"stat.ME","submitted_at":"2023-11-23T22:41:30Z","cross_cats_sorted":["cs.LG","stat.ML"],"title_canon_sha256":"1f96c5080c1dd533da22aa41124d4469a8f0457e54ccf6e845eccee75de9815a","abstract_canon_sha256":"fcba8d05eeb18107f0ac9f998055814c665d076eeb47b335197b3705cd179b02"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:07:25.303062Z","signature_b64":"32nU7IHr73iSiH9y+ZVuikFrQG1kRex5rJmfBXXzyIVYJVtf6nS9exxBCUu/Ux7ocCtGdM1tGqUtg7AcuhspCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f1b7c09c7ab396f8881803cf52799a31b584f06f7278ef7316ffcc82a7eeb0ae","last_reissued_at":"2026-07-05T09:07:25.302499Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:07:25.302499Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Assumption-Lean and Data-Adaptive Post-Prediction Inference","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"stat.ME","authors_text":"Jiacheng Miao, Jiwei Zhao, Qiongshi Lu, Xinran Miao, Yixuan Wu","submitted_at":"2023-11-23T22:41:30Z","abstract_excerpt":"A primary challenge facing modern scientific research is the limited availability of gold-standard data which can be costly, labor-intensive, or invasive to obtain. With the rapid development of machine learning (ML), scientists can now employ ML algorithms to predict gold-standard outcomes with variables that are easier to obtain. However, these predicted outcomes are often used directly in subsequent statistical analyses, ignoring imprecision and heterogeneity introduced by the prediction procedure. This will likely result in false positive findings and invalid scientific conclusions. In thi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.14220","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.14220/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.14220","created_at":"2026-07-05T09:07:25.302574+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.14220v4","created_at":"2026-07-05T09:07:25.302574+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.14220","created_at":"2026-07-05T09:07:25.302574+00:00"},{"alias_kind":"pith_short_12","alias_value":"6G34BHD2WOLP","created_at":"2026-07-05T09:07:25.302574+00:00"},{"alias_kind":"pith_short_16","alias_value":"6G34BHD2WOLPRCAY","created_at":"2026-07-05T09:07:25.302574+00:00"},{"alias_kind":"pith_short_8","alias_value":"6G34BHD2","created_at":"2026-07-05T09:07:25.302574+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.05076","citing_title":"High-Dimensional Statistics: Reflections on Progress and Open Problems","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2403.03208","citing_title":"Active Statistical Inference","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2505.06452","citing_title":"Semiparametric semi-supervised learning for general targets under distribution shift and decaying overlap","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05076","citing_title":"High-Dimensional Statistics: Reflections on Progress and Open Problems","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18569","citing_title":"Revisiting Active Sequential Prediction-Powered Mean Estimation","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6G34BHD2WOLPRCAYAPHVE6M2GG","json":"https://pith.science/pith/6G34BHD2WOLPRCAYAPHVE6M2GG.json","graph_json":"https://pith.science/api/pith-number/6G34BHD2WOLPRCAYAPHVE6M2GG/graph.json","events_json":"https://pith.science/api/pith-number/6G34BHD2WOLPRCAYAPHVE6M2GG/events.json","paper":"https://pith.science/paper/6G34BHD2"},"agent_actions":{"view_html":"https://pith.science/pith/6G34BHD2WOLPRCAYAPHVE6M2GG","download_json":"https://pith.science/pith/6G34BHD2WOLPRCAYAPHVE6M2GG.json","view_paper":"https://pith.science/paper/6G34BHD2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.14220&json=true","fetch_graph":"https://pith.science/api/pith-number/6G34BHD2WOLPRCAYAPHVE6M2GG/graph.json","fetch_events":"https://pith.science/api/pith-number/6G34BHD2WOLPRCAYAPHVE6M2GG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6G34BHD2WOLPRCAYAPHVE6M2GG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6G34BHD2WOLPRCAYAPHVE6M2GG/action/storage_attestation","attest_author":"https://pith.science/pith/6G34BHD2WOLPRCAYAPHVE6M2GG/action/author_attestation","sign_citation":"https://pith.science/pith/6G34BHD2WOLPRCAYAPHVE6M2GG/action/citation_signature","submit_replication":"https://pith.science/pith/6G34BHD2WOLPRCAYAPHVE6M2GG/action/replication_record"}},"created_at":"2026-07-05T09:07:25.302574+00:00","updated_at":"2026-07-05T09:07:25.302574+00:00"}