{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:3WGQURRVXWUHYNH73WDXQSZQUQ","short_pith_number":"pith:3WGQURRV","schema_version":"1.0","canonical_sha256":"dd8d0a4635bda87c34ffdd87784b30a41b7b79ec830b9fc7bea66f836f6436f9","source":{"kind":"arxiv","id":"2202.01327","version":1},"attestation_state":"computed","paper":{"title":"Adaptive Sampling Strategies to Construct Equitable Training Datasets","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","stat.ME"],"primary_cat":"cs.LG","authors_text":"Bobbie Chern, Miranda Bogen, Ro Encarnacion, Sam Corbett-Davies, Sharad Goel, Stevie Bergman, William Cai","submitted_at":"2022-01-31T19:19:30Z","abstract_excerpt":"In domains ranging from computer vision to natural language processing, machine learning models have been shown to exhibit stark disparities, often performing worse for members of traditionally underserved groups. One factor contributing to these performance gaps is a lack of representation in the data the models are trained on. It is often unclear, however, how to operationalize representativeness in specific applications. Here we formalize the problem of creating equitable training datasets, and propose a statistical framework for addressing this problem. We consider a setting where a model "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2202.01327","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-01-31T19:19:30Z","cross_cats_sorted":["cs.AI","stat.ME"],"title_canon_sha256":"4f0ce3fe29f88976a15b70c9b2b47077548012cb886cf69c989e34b2deb9a510","abstract_canon_sha256":"de50de5f821123d02fa3d17079f53eb51b0b0c7871b17b8abda8f2ba713b0e63"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:53:53.050494Z","signature_b64":"jAvImaHnaBNbwrnwlBfETDcTciz32ql4CGdVRBN4UIQ4KpBe1vb1XOcS/qOZ9nNte+A6eEGvoudrOsOd9qpVAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dd8d0a4635bda87c34ffdd87784b30a41b7b79ec830b9fc7bea66f836f6436f9","last_reissued_at":"2026-07-05T03:53:53.050081Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:53:53.050081Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Adaptive Sampling Strategies to Construct Equitable Training Datasets","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","stat.ME"],"primary_cat":"cs.LG","authors_text":"Bobbie Chern, Miranda Bogen, Ro Encarnacion, Sam Corbett-Davies, Sharad Goel, Stevie Bergman, William Cai","submitted_at":"2022-01-31T19:19:30Z","abstract_excerpt":"In domains ranging from computer vision to natural language processing, machine learning models have been shown to exhibit stark disparities, often performing worse for members of traditionally underserved groups. One factor contributing to these performance gaps is a lack of representation in the data the models are trained on. It is often unclear, however, how to operationalize representativeness in specific applications. Here we formalize the problem of creating equitable training datasets, and propose a statistical framework for addressing this problem. We consider a setting where a model "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2202.01327","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2202.01327/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2202.01327","created_at":"2026-07-05T03:53:53.050141+00:00"},{"alias_kind":"arxiv_version","alias_value":"2202.01327v1","created_at":"2026-07-05T03:53:53.050141+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2202.01327","created_at":"2026-07-05T03:53:53.050141+00:00"},{"alias_kind":"pith_short_12","alias_value":"3WGQURRVXWUH","created_at":"2026-07-05T03:53:53.050141+00:00"},{"alias_kind":"pith_short_16","alias_value":"3WGQURRVXWUHYNH7","created_at":"2026-07-05T03:53:53.050141+00:00"},{"alias_kind":"pith_short_8","alias_value":"3WGQURRV","created_at":"2026-07-05T03:53:53.050141+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3WGQURRVXWUHYNH73WDXQSZQUQ","json":"https://pith.science/pith/3WGQURRVXWUHYNH73WDXQSZQUQ.json","graph_json":"https://pith.science/api/pith-number/3WGQURRVXWUHYNH73WDXQSZQUQ/graph.json","events_json":"https://pith.science/api/pith-number/3WGQURRVXWUHYNH73WDXQSZQUQ/events.json","paper":"https://pith.science/paper/3WGQURRV"},"agent_actions":{"view_html":"https://pith.science/pith/3WGQURRVXWUHYNH73WDXQSZQUQ","download_json":"https://pith.science/pith/3WGQURRVXWUHYNH73WDXQSZQUQ.json","view_paper":"https://pith.science/paper/3WGQURRV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2202.01327&json=true","fetch_graph":"https://pith.science/api/pith-number/3WGQURRVXWUHYNH73WDXQSZQUQ/graph.json","fetch_events":"https://pith.science/api/pith-number/3WGQURRVXWUHYNH73WDXQSZQUQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3WGQURRVXWUHYNH73WDXQSZQUQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3WGQURRVXWUHYNH73WDXQSZQUQ/action/storage_attestation","attest_author":"https://pith.science/pith/3WGQURRVXWUHYNH73WDXQSZQUQ/action/author_attestation","sign_citation":"https://pith.science/pith/3WGQURRVXWUHYNH73WDXQSZQUQ/action/citation_signature","submit_replication":"https://pith.science/pith/3WGQURRVXWUHYNH73WDXQSZQUQ/action/replication_record"}},"created_at":"2026-07-05T03:53:53.050141+00:00","updated_at":"2026-07-05T03:53:53.050141+00:00"}