{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:ZHVCFC6G2WCDYRZU72YMYQTKIX","short_pith_number":"pith:ZHVCFC6G","schema_version":"1.0","canonical_sha256":"c9ea228bc6d5843c4734feb0cc426a45d6b6b12c45bcd9b3f10ce7e12e551a2a","source":{"kind":"arxiv","id":"1906.11829","version":4},"attestation_state":"computed","paper":{"title":"Selection via Proxy: Efficient Data Selection for Deep Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Baharan Mirzasoleiman, Christopher Yeh, Cody Coleman, Jure Leskovec, Matei Zaharia, Percy Liang, Peter Bailis, Stephen Mussmann","submitted_at":"2019-06-26T23:01:47Z","abstract_excerpt":"Data selection methods, such as active learning and core-set selection, are useful tools for machine learning on large datasets. However, they can be prohibitively expensive to apply in deep learning because they depend on feature representations that need to be learned. In this work, we show that we can greatly improve the computational efficiency by using a small proxy model to perform data selection (e.g., selecting data points to label for active learning). By removing hidden layers from the target model, using smaller architectures, and training for fewer epochs, we create proxies that ar"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1906.11829","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-06-26T23:01:47Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"f0c3f6c5f073f420e70688524e48e41dba2f38878d7b626ee3826b47732eafe2","abstract_canon_sha256":"80115dcbd8756ce5fc8b0af3c553a3f20c1d62c1c4cd01b89506c36ebe27a74d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:46:11.538764Z","signature_b64":"Lz7gLz2oSr9fkvqlVEatwEQdrymgYRqGR4kEJ67O4pWn1LZ0+5XWVj++ZgFqmD1yVOwYku1gNViLBBKUTxmhBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c9ea228bc6d5843c4734feb0cc426a45d6b6b12c45bcd9b3f10ce7e12e551a2a","last_reissued_at":"2026-07-05T01:46:11.538196Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:46:11.538196Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Selection via Proxy: Efficient Data Selection for Deep Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Baharan Mirzasoleiman, Christopher Yeh, Cody Coleman, Jure Leskovec, Matei Zaharia, Percy Liang, Peter Bailis, Stephen Mussmann","submitted_at":"2019-06-26T23:01:47Z","abstract_excerpt":"Data selection methods, such as active learning and core-set selection, are useful tools for machine learning on large datasets. However, they can be prohibitively expensive to apply in deep learning because they depend on feature representations that need to be learned. In this work, we show that we can greatly improve the computational efficiency by using a small proxy model to perform data selection (e.g., selecting data points to label for active learning). By removing hidden layers from the target model, using smaller architectures, and training for fewer epochs, we create proxies that ar"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1906.11829","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1906.11829/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1906.11829","created_at":"2026-07-05T01:46:11.538260+00:00"},{"alias_kind":"arxiv_version","alias_value":"1906.11829v4","created_at":"2026-07-05T01:46:11.538260+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1906.11829","created_at":"2026-07-05T01:46:11.538260+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZHVCFC6G2WCD","created_at":"2026-07-05T01:46:11.538260+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZHVCFC6G2WCDYRZU","created_at":"2026-07-05T01:46:11.538260+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZHVCFC6G","created_at":"2026-07-05T01:46:11.538260+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18209","citing_title":"Rethinking Dataset Distillation for Classification: Do Distilled Sets Outperform Coresets?","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08398","citing_title":"Exploring and Exploiting Stability in Latent Flow Matching","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23482","citing_title":"Multimodal Distribution Matching for Vision-Language Dataset Distillation","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2502.12272","citing_title":"Learning to Reason at the Frontier of Learnability","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2502.10248","citing_title":"Step-Video-T2V Technical Report: The Practice, Challenges, and Future of Video Foundation Model","ref_index":206,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08398","citing_title":"Exploring and Exploiting Stability in Latent Flow Matching","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11328","citing_title":"Select Smarter, Not More: Prompt-Aware Evaluation Scheduling with Submodular Guarantees","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07306","citing_title":"Beyond Loss Values: Robust Dynamic Pruning via Loss Trajectory Alignment","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZHVCFC6G2WCDYRZU72YMYQTKIX","json":"https://pith.science/pith/ZHVCFC6G2WCDYRZU72YMYQTKIX.json","graph_json":"https://pith.science/api/pith-number/ZHVCFC6G2WCDYRZU72YMYQTKIX/graph.json","events_json":"https://pith.science/api/pith-number/ZHVCFC6G2WCDYRZU72YMYQTKIX/events.json","paper":"https://pith.science/paper/ZHVCFC6G"},"agent_actions":{"view_html":"https://pith.science/pith/ZHVCFC6G2WCDYRZU72YMYQTKIX","download_json":"https://pith.science/pith/ZHVCFC6G2WCDYRZU72YMYQTKIX.json","view_paper":"https://pith.science/paper/ZHVCFC6G","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1906.11829&json=true","fetch_graph":"https://pith.science/api/pith-number/ZHVCFC6G2WCDYRZU72YMYQTKIX/graph.json","fetch_events":"https://pith.science/api/pith-number/ZHVCFC6G2WCDYRZU72YMYQTKIX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZHVCFC6G2WCDYRZU72YMYQTKIX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZHVCFC6G2WCDYRZU72YMYQTKIX/action/storage_attestation","attest_author":"https://pith.science/pith/ZHVCFC6G2WCDYRZU72YMYQTKIX/action/author_attestation","sign_citation":"https://pith.science/pith/ZHVCFC6G2WCDYRZU72YMYQTKIX/action/citation_signature","submit_replication":"https://pith.science/pith/ZHVCFC6G2WCDYRZU72YMYQTKIX/action/replication_record"}},"created_at":"2026-07-05T01:46:11.538260+00:00","updated_at":"2026-07-05T01:46:11.538260+00:00"}