{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:5ZFXRQRYANQQZTJTA3U54IA2CQ","short_pith_number":"pith:5ZFXRQRY","schema_version":"1.0","canonical_sha256":"ee4b78c23803610ccd3306e9de201a143f829442113f417f1526da5603cbadd3","source":{"kind":"arxiv","id":"2204.13749","version":2},"attestation_state":"computed","paper":{"title":"Learning to Split for Automatic Bias Detection","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CV"],"primary_cat":"cs.LG","authors_text":"Regina Barzilay, Yujia Bao","submitted_at":"2022-04-28T19:41:08Z","abstract_excerpt":"Classifiers are biased when trained on biased datasets. As a remedy, we propose Learning to Split (ls), an algorithm for automatic bias detection. Given a dataset with input-label pairs, ls learns to split this dataset so that predictors trained on the training split cannot generalize to the testing split. This performance gap suggests that the testing split is under-represented in the dataset, which is a signal of potential bias. Identifying non-generalizable splits is challenging since we have no annotations about the bias. In this work, we show that the prediction correctness of each exampl"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2204.13749","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-04-28T19:41:08Z","cross_cats_sorted":["cs.AI","cs.CL","cs.CV"],"title_canon_sha256":"153c41bf7ecc2e9e0934fc7cef598c6a7cdfc68703e19ad7328cb1662ac81d36","abstract_canon_sha256":"77a57f4f1cb84b4637c1a48cd9f420517a842e4f267c5ff9295f3540e62b9598"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:42:18.891996Z","signature_b64":"6npxjoUjpGhRiOOS8Lg/cuAviErexfNT5ONELWhCrPU5SLO64TfSPPSWjNy7vGoUOEPsjLaLcGt/tgLX/81UCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ee4b78c23803610ccd3306e9de201a143f829442113f417f1526da5603cbadd3","last_reissued_at":"2026-07-05T04:42:18.891521Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:42:18.891521Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning to Split for Automatic Bias Detection","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CV"],"primary_cat":"cs.LG","authors_text":"Regina Barzilay, Yujia Bao","submitted_at":"2022-04-28T19:41:08Z","abstract_excerpt":"Classifiers are biased when trained on biased datasets. As a remedy, we propose Learning to Split (ls), an algorithm for automatic bias detection. Given a dataset with input-label pairs, ls learns to split this dataset so that predictors trained on the training split cannot generalize to the testing split. This performance gap suggests that the testing split is under-represented in the dataset, which is a signal of potential bias. Identifying non-generalizable splits is challenging since we have no annotations about the bias. In this work, we show that the prediction correctness of each exampl"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2204.13749","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2204.13749/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2204.13749","created_at":"2026-07-05T04:42:18.891578+00:00"},{"alias_kind":"arxiv_version","alias_value":"2204.13749v2","created_at":"2026-07-05T04:42:18.891578+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2204.13749","created_at":"2026-07-05T04:42:18.891578+00:00"},{"alias_kind":"pith_short_12","alias_value":"5ZFXRQRYANQQ","created_at":"2026-07-05T04:42:18.891578+00:00"},{"alias_kind":"pith_short_16","alias_value":"5ZFXRQRYANQQZTJT","created_at":"2026-07-05T04:42:18.891578+00:00"},{"alias_kind":"pith_short_8","alias_value":"5ZFXRQRY","created_at":"2026-07-05T04:42:18.891578+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2501.06669","citing_title":"Challenging reaction prediction models to generalize to novel chemistry","ref_index":3,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5ZFXRQRYANQQZTJTA3U54IA2CQ","json":"https://pith.science/pith/5ZFXRQRYANQQZTJTA3U54IA2CQ.json","graph_json":"https://pith.science/api/pith-number/5ZFXRQRYANQQZTJTA3U54IA2CQ/graph.json","events_json":"https://pith.science/api/pith-number/5ZFXRQRYANQQZTJTA3U54IA2CQ/events.json","paper":"https://pith.science/paper/5ZFXRQRY"},"agent_actions":{"view_html":"https://pith.science/pith/5ZFXRQRYANQQZTJTA3U54IA2CQ","download_json":"https://pith.science/pith/5ZFXRQRYANQQZTJTA3U54IA2CQ.json","view_paper":"https://pith.science/paper/5ZFXRQRY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2204.13749&json=true","fetch_graph":"https://pith.science/api/pith-number/5ZFXRQRYANQQZTJTA3U54IA2CQ/graph.json","fetch_events":"https://pith.science/api/pith-number/5ZFXRQRYANQQZTJTA3U54IA2CQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5ZFXRQRYANQQZTJTA3U54IA2CQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5ZFXRQRYANQQZTJTA3U54IA2CQ/action/storage_attestation","attest_author":"https://pith.science/pith/5ZFXRQRYANQQZTJTA3U54IA2CQ/action/author_attestation","sign_citation":"https://pith.science/pith/5ZFXRQRYANQQZTJTA3U54IA2CQ/action/citation_signature","submit_replication":"https://pith.science/pith/5ZFXRQRYANQQZTJTA3U54IA2CQ/action/replication_record"}},"created_at":"2026-07-05T04:42:18.891578+00:00","updated_at":"2026-07-05T04:42:18.891578+00:00"}