{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:FDWALGHBQUFEDGVNYMLJ7O3MQ5","short_pith_number":"pith:FDWALGHB","schema_version":"1.0","canonical_sha256":"28ec0598e1850a419aadc3169fbb6c8750282bde0847fffb124b29e67cc28746","source":{"kind":"arxiv","id":"2103.05331","version":2},"attestation_state":"computed","paper":{"title":"Active Testing: Sample-Efficient Model Evaluation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Jannik Kossen, Sebastian Farquhar, Tom Rainforth, Yarin Gal","submitted_at":"2021-03-09T10:20:49Z","abstract_excerpt":"We introduce a new framework for sample-efficient model evaluation that we call active testing. While approaches like active learning reduce the number of labels needed for model training, existing literature largely ignores the cost of labeling test data, typically unrealistically assuming large test sets for model evaluation. This creates a disconnect to real applications, where test labels are important and just as expensive, e.g. for optimizing hyperparameters. Active testing addresses this by carefully selecting the test points to label, ensuring model evaluation is sample-efficient. To t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2103.05331","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"stat.ML","submitted_at":"2021-03-09T10:20:49Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"2a6f0e88ce2f27afc64d34ada52add145f7c5b15d4a28b715f65e7af3c6ae388","abstract_canon_sha256":"c3f65fe6ddb20cd36423b49d483ef5e4a2bbed32953bcedc3339ec45de70fa3a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:49:00.040174Z","signature_b64":"3eLuPuqLccoB2/62KfORm6OpV+8I7yOmwqnfYYXPPGUg1/WwExHW0dkMRtlDtabi3HJDCGms06oQoE98TjGaCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"28ec0598e1850a419aadc3169fbb6c8750282bde0847fffb124b29e67cc28746","last_reissued_at":"2026-07-05T02:49:00.039708Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:49:00.039708Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Active Testing: Sample-Efficient Model Evaluation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Jannik Kossen, Sebastian Farquhar, Tom Rainforth, Yarin Gal","submitted_at":"2021-03-09T10:20:49Z","abstract_excerpt":"We introduce a new framework for sample-efficient model evaluation that we call active testing. While approaches like active learning reduce the number of labels needed for model training, existing literature largely ignores the cost of labeling test data, typically unrealistically assuming large test sets for model evaluation. This creates a disconnect to real applications, where test labels are important and just as expensive, e.g. for optimizing hyperparameters. Active testing addresses this by carefully selecting the test points to label, ensuring model evaluation is sample-efficient. To t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2103.05331","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2103.05331/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2103.05331","created_at":"2026-07-05T02:49:00.039763+00:00"},{"alias_kind":"arxiv_version","alias_value":"2103.05331v2","created_at":"2026-07-05T02:49:00.039763+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2103.05331","created_at":"2026-07-05T02:49:00.039763+00:00"},{"alias_kind":"pith_short_12","alias_value":"FDWALGHBQUFE","created_at":"2026-07-05T02:49:00.039763+00:00"},{"alias_kind":"pith_short_16","alias_value":"FDWALGHBQUFEDGVN","created_at":"2026-07-05T02:49:00.039763+00:00"},{"alias_kind":"pith_short_8","alias_value":"FDWALGHB","created_at":"2026-07-05T02:49:00.039763+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.27446","citing_title":"Causal Connections: Leveraging Multilingual Fine-Tuning for Financial QA@FinCausal 2026","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27316","citing_title":"LLM-Based Examination of Eligibility Criteria from Securities Prospectuses at the German Central Bank","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FDWALGHBQUFEDGVNYMLJ7O3MQ5","json":"https://pith.science/pith/FDWALGHBQUFEDGVNYMLJ7O3MQ5.json","graph_json":"https://pith.science/api/pith-number/FDWALGHBQUFEDGVNYMLJ7O3MQ5/graph.json","events_json":"https://pith.science/api/pith-number/FDWALGHBQUFEDGVNYMLJ7O3MQ5/events.json","paper":"https://pith.science/paper/FDWALGHB"},"agent_actions":{"view_html":"https://pith.science/pith/FDWALGHBQUFEDGVNYMLJ7O3MQ5","download_json":"https://pith.science/pith/FDWALGHBQUFEDGVNYMLJ7O3MQ5.json","view_paper":"https://pith.science/paper/FDWALGHB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2103.05331&json=true","fetch_graph":"https://pith.science/api/pith-number/FDWALGHBQUFEDGVNYMLJ7O3MQ5/graph.json","fetch_events":"https://pith.science/api/pith-number/FDWALGHBQUFEDGVNYMLJ7O3MQ5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FDWALGHBQUFEDGVNYMLJ7O3MQ5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FDWALGHBQUFEDGVNYMLJ7O3MQ5/action/storage_attestation","attest_author":"https://pith.science/pith/FDWALGHBQUFEDGVNYMLJ7O3MQ5/action/author_attestation","sign_citation":"https://pith.science/pith/FDWALGHBQUFEDGVNYMLJ7O3MQ5/action/citation_signature","submit_replication":"https://pith.science/pith/FDWALGHBQUFEDGVNYMLJ7O3MQ5/action/replication_record"}},"created_at":"2026-07-05T02:49:00.039763+00:00","updated_at":"2026-07-05T02:49:00.039763+00:00"}