{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:HXGEBUUXL65QSQTX4RBO5YZSQS","short_pith_number":"pith:HXGEBUUX","schema_version":"1.0","canonical_sha256":"3dcc40d2975fbb094277e442eee33284828a227d81696175b09cd5d5979b5854","source":{"kind":"arxiv","id":"2101.11665","version":2},"attestation_state":"computed","paper":{"title":"On Statistical Bias In Active Learning: How and When To Fix It","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Sebastian Farquhar, Tom Rainforth, Yarin Gal","submitted_at":"2021-01-27T19:52:24Z","abstract_excerpt":"Active learning is a powerful tool when labelling data is expensive, but it introduces a bias because the training data no longer follows the population distribution. We formalize this bias and investigate the situations in which it can be harmful and sometimes even helpful. We further introduce novel corrective weights to remove bias when doing so is beneficial. Through this, our work not only provides a useful mechanism that can improve the active learning approach, but also an explanation of the empirical successes of various existing approaches which ignore this bias. In particular, we sho"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2101.11665","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"stat.ML","submitted_at":"2021-01-27T19:52:24Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"590a539554e85dd248b0691a26e787d3ef0dfa2247cc86633bf2020483f0c2ae","abstract_canon_sha256":"1a638b94030e7a2a44060dcf32013e8bf9a711fd1be96ef5bb8cfb47fbda3fa5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:44:35.541803Z","signature_b64":"MUX5yfLhxOw5jyzu5PgFUZjJW2PTRex5Y14au+QC0QXJy0QWnvzyvObd56xq92wFlByKFhcVcV0uezJV7pW+AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3dcc40d2975fbb094277e442eee33284828a227d81696175b09cd5d5979b5854","last_reissued_at":"2026-07-05T02:44:35.541400Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:44:35.541400Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On Statistical Bias In Active Learning: How and When To Fix It","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Sebastian Farquhar, Tom Rainforth, Yarin Gal","submitted_at":"2021-01-27T19:52:24Z","abstract_excerpt":"Active learning is a powerful tool when labelling data is expensive, but it introduces a bias because the training data no longer follows the population distribution. We formalize this bias and investigate the situations in which it can be harmful and sometimes even helpful. We further introduce novel corrective weights to remove bias when doing so is beneficial. Through this, our work not only provides a useful mechanism that can improve the active learning approach, but also an explanation of the empirical successes of various existing approaches which ignore this bias. In particular, we sho"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2101.11665","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2101.11665/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2101.11665","created_at":"2026-07-05T02:44:35.541459+00:00"},{"alias_kind":"arxiv_version","alias_value":"2101.11665v2","created_at":"2026-07-05T02:44:35.541459+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2101.11665","created_at":"2026-07-05T02:44:35.541459+00:00"},{"alias_kind":"pith_short_12","alias_value":"HXGEBUUXL65Q","created_at":"2026-07-05T02:44:35.541459+00:00"},{"alias_kind":"pith_short_16","alias_value":"HXGEBUUXL65QSQTX","created_at":"2026-07-05T02:44:35.541459+00:00"},{"alias_kind":"pith_short_8","alias_value":"HXGEBUUX","created_at":"2026-07-05T02:44:35.541459+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08347","citing_title":"Prediction-Powered Active Testing","ref_index":8,"is_internal_anchor":true},{"citing_arxiv_id":"2606.27446","citing_title":"Causal Connections: Leveraging Multilingual Fine-Tuning for Financial QA@FinCausal 2026","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27316","citing_title":"LLM-Based Examination of Eligibility Criteria from Securities Prospectuses at the German Central Bank","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10075","citing_title":"Active Testing of Large Language Models via Approximate Neyman Allocation","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10075","citing_title":"Active Testing of Large Language Models via Approximate Neyman Allocation","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01687","citing_title":"MultiBreak: A Scalable and Diverse Multi-turn Jailbreak Benchmark for Evaluating LLM Safety","ref_index":45,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HXGEBUUXL65QSQTX4RBO5YZSQS","json":"https://pith.science/pith/HXGEBUUXL65QSQTX4RBO5YZSQS.json","graph_json":"https://pith.science/api/pith-number/HXGEBUUXL65QSQTX4RBO5YZSQS/graph.json","events_json":"https://pith.science/api/pith-number/HXGEBUUXL65QSQTX4RBO5YZSQS/events.json","paper":"https://pith.science/paper/HXGEBUUX"},"agent_actions":{"view_html":"https://pith.science/pith/HXGEBUUXL65QSQTX4RBO5YZSQS","download_json":"https://pith.science/pith/HXGEBUUXL65QSQTX4RBO5YZSQS.json","view_paper":"https://pith.science/paper/HXGEBUUX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2101.11665&json=true","fetch_graph":"https://pith.science/api/pith-number/HXGEBUUXL65QSQTX4RBO5YZSQS/graph.json","fetch_events":"https://pith.science/api/pith-number/HXGEBUUXL65QSQTX4RBO5YZSQS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HXGEBUUXL65QSQTX4RBO5YZSQS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HXGEBUUXL65QSQTX4RBO5YZSQS/action/storage_attestation","attest_author":"https://pith.science/pith/HXGEBUUXL65QSQTX4RBO5YZSQS/action/author_attestation","sign_citation":"https://pith.science/pith/HXGEBUUXL65QSQTX4RBO5YZSQS/action/citation_signature","submit_replication":"https://pith.science/pith/HXGEBUUXL65QSQTX4RBO5YZSQS/action/replication_record"}},"created_at":"2026-07-05T02:44:35.541459+00:00","updated_at":"2026-07-05T02:44:35.541459+00:00"}