{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:UE5DPBYJTZ36AAAWJLLJFOQWIH","short_pith_number":"pith:UE5DPBYJ","schema_version":"1.0","canonical_sha256":"a13a3787099e77e000164ad692ba1641db8b090a52cba18797e7a8956e035d5d","source":{"kind":"arxiv","id":"2009.13447","version":3},"attestation_state":"computed","paper":{"title":"Why resampling outperforms reweighting for correcting sampling bias with stochastic gradients","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.NA","math.NA","stat.ML"],"primary_cat":"cs.LG","authors_text":"Jing An, Lexing Ying, Yuhua Zhu","submitted_at":"2020-09-28T16:12:38Z","abstract_excerpt":"A data set sampled from a certain population is biased if the subgroups of the population are sampled at proportions that are significantly different from their underlying proportions. Training machine learning models on biased data sets requires correction techniques to compensate for the bias. We consider two commonly-used techniques, resampling and reweighting, that rebalance the proportions of the subgroups to maintain the desired objective function. Though statistically equivalent, it has been observed that resampling outperforms reweighting when combined with stochastic gradient algorith"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2009.13447","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-09-28T16:12:38Z","cross_cats_sorted":["cs.NA","math.NA","stat.ML"],"title_canon_sha256":"65ec90dce2a5e949c44129777a117d5f5731997f65107f2c7fa78b5fffd35dec","abstract_canon_sha256":"c5d9347a2a67dc7ddc39433172afcb0e15fb939fe00af51cfade1a104ec59efb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:09:15.998952Z","signature_b64":"zGb3C2g0DHwv8P3rWjzjGBFJ8VzhnaOAcvN/QEmp4h/7bbEjoh5WHZsf9m6LYezOPIyuxL+5LcKNM5L5pwlxAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a13a3787099e77e000164ad692ba1641db8b090a52cba18797e7a8956e035d5d","last_reissued_at":"2026-07-05T03:09:15.998570Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:09:15.998570Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Why resampling outperforms reweighting for correcting sampling bias with stochastic gradients","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.NA","math.NA","stat.ML"],"primary_cat":"cs.LG","authors_text":"Jing An, Lexing Ying, Yuhua Zhu","submitted_at":"2020-09-28T16:12:38Z","abstract_excerpt":"A data set sampled from a certain population is biased if the subgroups of the population are sampled at proportions that are significantly different from their underlying proportions. Training machine learning models on biased data sets requires correction techniques to compensate for the bias. We consider two commonly-used techniques, resampling and reweighting, that rebalance the proportions of the subgroups to maintain the desired objective function. Though statistically equivalent, it has been observed that resampling outperforms reweighting when combined with stochastic gradient algorith"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2009.13447","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2009.13447/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2009.13447","created_at":"2026-07-05T03:09:15.998624+00:00"},{"alias_kind":"arxiv_version","alias_value":"2009.13447v3","created_at":"2026-07-05T03:09:15.998624+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2009.13447","created_at":"2026-07-05T03:09:15.998624+00:00"},{"alias_kind":"pith_short_12","alias_value":"UE5DPBYJTZ36","created_at":"2026-07-05T03:09:15.998624+00:00"},{"alias_kind":"pith_short_16","alias_value":"UE5DPBYJTZ36AAAW","created_at":"2026-07-05T03:09:15.998624+00:00"},{"alias_kind":"pith_short_8","alias_value":"UE5DPBYJ","created_at":"2026-07-05T03:09:15.998624+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.14350","citing_title":"Distributionally Robust Multi-Task Reinforcement Learning via Adaptive Task Sampling","ref_index":243,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UE5DPBYJTZ36AAAWJLLJFOQWIH","json":"https://pith.science/pith/UE5DPBYJTZ36AAAWJLLJFOQWIH.json","graph_json":"https://pith.science/api/pith-number/UE5DPBYJTZ36AAAWJLLJFOQWIH/graph.json","events_json":"https://pith.science/api/pith-number/UE5DPBYJTZ36AAAWJLLJFOQWIH/events.json","paper":"https://pith.science/paper/UE5DPBYJ"},"agent_actions":{"view_html":"https://pith.science/pith/UE5DPBYJTZ36AAAWJLLJFOQWIH","download_json":"https://pith.science/pith/UE5DPBYJTZ36AAAWJLLJFOQWIH.json","view_paper":"https://pith.science/paper/UE5DPBYJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2009.13447&json=true","fetch_graph":"https://pith.science/api/pith-number/UE5DPBYJTZ36AAAWJLLJFOQWIH/graph.json","fetch_events":"https://pith.science/api/pith-number/UE5DPBYJTZ36AAAWJLLJFOQWIH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UE5DPBYJTZ36AAAWJLLJFOQWIH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UE5DPBYJTZ36AAAWJLLJFOQWIH/action/storage_attestation","attest_author":"https://pith.science/pith/UE5DPBYJTZ36AAAWJLLJFOQWIH/action/author_attestation","sign_citation":"https://pith.science/pith/UE5DPBYJTZ36AAAWJLLJFOQWIH/action/citation_signature","submit_replication":"https://pith.science/pith/UE5DPBYJTZ36AAAWJLLJFOQWIH/action/replication_record"}},"created_at":"2026-07-05T03:09:15.998624+00:00","updated_at":"2026-07-05T03:09:15.998624+00:00"}