{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:7NO6S3ZZQZ23WWHF2UURONM6AS","short_pith_number":"pith:7NO6S3ZZ","schema_version":"1.0","canonical_sha256":"fb5de96f398675bb58e5d52917359e04adb068d3d0056b18ee5ae22058e706c8","source":{"kind":"arxiv","id":"2205.03246","version":2},"attestation_state":"computed","paper":{"title":"What Makes A Good Fisherman? Linear Regression under Self-Selection Bias","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.DS","cs.LG","stat.ML","stat.TH"],"primary_cat":"math.ST","authors_text":"Andrew Ilyas, Constantinos Daskalakis, Manolis Zampetakis, Yeshwanth Cherapanamjeri","submitted_at":"2022-05-06T14:03:05Z","abstract_excerpt":"In the classical setting of self-selection, the goal is to learn $k$ models, simultaneously from observations $(x^{(i)}, y^{(i)})$ where $y^{(i)}$ is the output of one of $k$ underlying models on input $x^{(i)}$. In contrast to mixture models, where we observe the output of a randomly selected model, here the observed model depends on the outputs themselves, and is determined by some known selection criterion. For example, we might observe the highest output, the smallest output, or the median output of the $k$ models. In known-index self-selection, the identity of the observed model output is"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2205.03246","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"math.ST","submitted_at":"2022-05-06T14:03:05Z","cross_cats_sorted":["cs.DS","cs.LG","stat.ML","stat.TH"],"title_canon_sha256":"b96fe2ee8b48c9a5fddf509a9d4ce82df06ddd8561b688fdd92d31fda2e3185d","abstract_canon_sha256":"0bebcd119e73ea546f35a0d8c240ea2170456047890798ac3ee91af0c21cfaa1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:23:57.769494Z","signature_b64":"tHbXUKSPsCo3QnICis+w4v9FJrKobpfq66RPmzpDbLYEROh/z1Vyz7hF1ReQuspG+sIXbWEjJjPosK34wNO2BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fb5de96f398675bb58e5d52917359e04adb068d3d0056b18ee5ae22058e706c8","last_reissued_at":"2026-07-05T05:23:57.769080Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:23:57.769080Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"What Makes A Good Fisherman? Linear Regression under Self-Selection Bias","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.DS","cs.LG","stat.ML","stat.TH"],"primary_cat":"math.ST","authors_text":"Andrew Ilyas, Constantinos Daskalakis, Manolis Zampetakis, Yeshwanth Cherapanamjeri","submitted_at":"2022-05-06T14:03:05Z","abstract_excerpt":"In the classical setting of self-selection, the goal is to learn $k$ models, simultaneously from observations $(x^{(i)}, y^{(i)})$ where $y^{(i)}$ is the output of one of $k$ underlying models on input $x^{(i)}$. In contrast to mixture models, where we observe the output of a randomly selected model, here the observed model depends on the outputs themselves, and is determined by some known selection criterion. For example, we might observe the highest output, the smallest output, or the median output of the $k$ models. In known-index self-selection, the identity of the observed model output is"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2205.03246","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2205.03246/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2205.03246","created_at":"2026-07-05T05:23:57.769135+00:00"},{"alias_kind":"arxiv_version","alias_value":"2205.03246v2","created_at":"2026-07-05T05:23:57.769135+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2205.03246","created_at":"2026-07-05T05:23:57.769135+00:00"},{"alias_kind":"pith_short_12","alias_value":"7NO6S3ZZQZ23","created_at":"2026-07-05T05:23:57.769135+00:00"},{"alias_kind":"pith_short_16","alias_value":"7NO6S3ZZQZ23WWHF","created_at":"2026-07-05T05:23:57.769135+00:00"},{"alias_kind":"pith_short_8","alias_value":"7NO6S3ZZ","created_at":"2026-07-05T05:23:57.769135+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2504.19446","citing_title":"Learning High-dimensional Gaussians from Censored Data","ref_index":18,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7NO6S3ZZQZ23WWHF2UURONM6AS","json":"https://pith.science/pith/7NO6S3ZZQZ23WWHF2UURONM6AS.json","graph_json":"https://pith.science/api/pith-number/7NO6S3ZZQZ23WWHF2UURONM6AS/graph.json","events_json":"https://pith.science/api/pith-number/7NO6S3ZZQZ23WWHF2UURONM6AS/events.json","paper":"https://pith.science/paper/7NO6S3ZZ"},"agent_actions":{"view_html":"https://pith.science/pith/7NO6S3ZZQZ23WWHF2UURONM6AS","download_json":"https://pith.science/pith/7NO6S3ZZQZ23WWHF2UURONM6AS.json","view_paper":"https://pith.science/paper/7NO6S3ZZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2205.03246&json=true","fetch_graph":"https://pith.science/api/pith-number/7NO6S3ZZQZ23WWHF2UURONM6AS/graph.json","fetch_events":"https://pith.science/api/pith-number/7NO6S3ZZQZ23WWHF2UURONM6AS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7NO6S3ZZQZ23WWHF2UURONM6AS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7NO6S3ZZQZ23WWHF2UURONM6AS/action/storage_attestation","attest_author":"https://pith.science/pith/7NO6S3ZZQZ23WWHF2UURONM6AS/action/author_attestation","sign_citation":"https://pith.science/pith/7NO6S3ZZQZ23WWHF2UURONM6AS/action/citation_signature","submit_replication":"https://pith.science/pith/7NO6S3ZZQZ23WWHF2UURONM6AS/action/replication_record"}},"created_at":"2026-07-05T05:23:57.769135+00:00","updated_at":"2026-07-05T05:23:57.769135+00:00"}