{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:EQIKKKDBMEHVYXFIVVMSGZZNQV","short_pith_number":"pith:EQIKKKDB","schema_version":"1.0","canonical_sha256":"2410a52861610f5c5ca8ad5923672d854164821c9828dd18da338e47cdfcaa35","source":{"kind":"arxiv","id":"2209.04569","version":1},"attestation_state":"computed","paper":{"title":"Nearly optimal capture-recapture sampling and empirical likelihood weighting estimation for M-estimation with big data","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"stat.ME","authors_text":"Jing Qin, Yan Fan, Yang Liu, Yukun Liu","submitted_at":"2022-09-10T01:58:24Z","abstract_excerpt":"Subsampling techniques can reduce the computational costs of processing big data. Practical subsampling plans typically involve initial uniform sampling and refined sampling. With a subsample, big data inferences are generally built on the inverse probability weighting (IPW), which becomes unstable when the probability weights are close to zero and cannot incorporate auxiliary information. First, we consider capture-recapture sampling, which combines an initial uniform sampling with a second Poisson sampling. Under this sampling plan, we propose an empirical likelihood weighting (ELW) estimati"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2209.04569","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"stat.ME","submitted_at":"2022-09-10T01:58:24Z","cross_cats_sorted":[],"title_canon_sha256":"b48ea656250a0b8034ca591ac181d99db31298d3e0178d8c9bf08553d2ef2c3b","abstract_canon_sha256":"27b194bdae75e9d51962ed90aa996137cc9cdfe03a023a5ef5e7068680f1a76d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:56:03.085914Z","signature_b64":"t9o/3p9rWpoyioviOliUcL2iB6AysCp22boeO1g3gPY2pHBVa58Dazi+WLbyKppAjxDkHkFYJdhfqY5DU6iCCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2410a52861610f5c5ca8ad5923672d854164821c9828dd18da338e47cdfcaa35","last_reissued_at":"2026-07-05T04:56:03.085500Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:56:03.085500Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Nearly optimal capture-recapture sampling and empirical likelihood weighting estimation for M-estimation with big data","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"stat.ME","authors_text":"Jing Qin, Yan Fan, Yang Liu, Yukun Liu","submitted_at":"2022-09-10T01:58:24Z","abstract_excerpt":"Subsampling techniques can reduce the computational costs of processing big data. Practical subsampling plans typically involve initial uniform sampling and refined sampling. With a subsample, big data inferences are generally built on the inverse probability weighting (IPW), which becomes unstable when the probability weights are close to zero and cannot incorporate auxiliary information. First, we consider capture-recapture sampling, which combines an initial uniform sampling with a second Poisson sampling. Under this sampling plan, we propose an empirical likelihood weighting (ELW) estimati"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2209.04569","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2209.04569/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2209.04569","created_at":"2026-07-05T04:56:03.085565+00:00"},{"alias_kind":"arxiv_version","alias_value":"2209.04569v1","created_at":"2026-07-05T04:56:03.085565+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2209.04569","created_at":"2026-07-05T04:56:03.085565+00:00"},{"alias_kind":"pith_short_12","alias_value":"EQIKKKDBMEHV","created_at":"2026-07-05T04:56:03.085565+00:00"},{"alias_kind":"pith_short_16","alias_value":"EQIKKKDBMEHVYXFI","created_at":"2026-07-05T04:56:03.085565+00:00"},{"alias_kind":"pith_short_8","alias_value":"EQIKKKDB","created_at":"2026-07-05T04:56:03.085565+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EQIKKKDBMEHVYXFIVVMSGZZNQV","json":"https://pith.science/pith/EQIKKKDBMEHVYXFIVVMSGZZNQV.json","graph_json":"https://pith.science/api/pith-number/EQIKKKDBMEHVYXFIVVMSGZZNQV/graph.json","events_json":"https://pith.science/api/pith-number/EQIKKKDBMEHVYXFIVVMSGZZNQV/events.json","paper":"https://pith.science/paper/EQIKKKDB"},"agent_actions":{"view_html":"https://pith.science/pith/EQIKKKDBMEHVYXFIVVMSGZZNQV","download_json":"https://pith.science/pith/EQIKKKDBMEHVYXFIVVMSGZZNQV.json","view_paper":"https://pith.science/paper/EQIKKKDB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2209.04569&json=true","fetch_graph":"https://pith.science/api/pith-number/EQIKKKDBMEHVYXFIVVMSGZZNQV/graph.json","fetch_events":"https://pith.science/api/pith-number/EQIKKKDBMEHVYXFIVVMSGZZNQV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EQIKKKDBMEHVYXFIVVMSGZZNQV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EQIKKKDBMEHVYXFIVVMSGZZNQV/action/storage_attestation","attest_author":"https://pith.science/pith/EQIKKKDBMEHVYXFIVVMSGZZNQV/action/author_attestation","sign_citation":"https://pith.science/pith/EQIKKKDBMEHVYXFIVVMSGZZNQV/action/citation_signature","submit_replication":"https://pith.science/pith/EQIKKKDBMEHVYXFIVVMSGZZNQV/action/replication_record"}},"created_at":"2026-07-05T04:56:03.085565+00:00","updated_at":"2026-07-05T04:56:03.085565+00:00"}