{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XV7YNAZNIVKDHJ26BCGXZUQROO","short_pith_number":"pith:XV7YNAZN","schema_version":"1.0","canonical_sha256":"bd7f86832d455433a75e088d7cd2117380012f387af1b22d768bf907c448d9c3","source":{"kind":"arxiv","id":"2404.05579","version":4},"attestation_state":"computed","paper":{"title":"DRoP: Distributionally Robust Data Pruning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.LG","authors_text":"Artem Vysogorets, Julia Kempe, Kartik Ahuja","submitted_at":"2024-04-08T14:55:35Z","abstract_excerpt":"In the era of exceptionally data-hungry models, careful selection of the training data is essential to mitigate the extensive costs of deep learning. Data pruning offers a solution by removing redundant or uninformative samples from the dataset, which yields faster convergence and improved neural scaling laws. However, little is known about its impact on classification bias of the trained models. We conduct the first systematic study of this effect and reveal that existing data pruning algorithms can produce highly biased classifiers. We present theoretical analysis of the classification risk "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.05579","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-04-08T14:55:35Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"5fd18421bff1da09cb46ce45ca2fc5e9feabc8ea4d89824c89578984628b3a32","abstract_canon_sha256":"e6d0206902d60acdf44a9d0e83236769c88bbc8f827e6a19690a0e64d327303a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:11:18.374301Z","signature_b64":"fWfq1R4iKltP6bgSc5x99NZI/4/BX9HIBUzdw5T6RKBbisCGM6QulSrMXpSjJf2IKDC1a4/YwFuLCU82yvSACA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bd7f86832d455433a75e088d7cd2117380012f387af1b22d768bf907c448d9c3","last_reissued_at":"2026-07-05T10:11:18.373793Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:11:18.373793Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DRoP: Distributionally Robust Data Pruning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.LG","authors_text":"Artem Vysogorets, Julia Kempe, Kartik Ahuja","submitted_at":"2024-04-08T14:55:35Z","abstract_excerpt":"In the era of exceptionally data-hungry models, careful selection of the training data is essential to mitigate the extensive costs of deep learning. Data pruning offers a solution by removing redundant or uninformative samples from the dataset, which yields faster convergence and improved neural scaling laws. However, little is known about its impact on classification bias of the trained models. We conduct the first systematic study of this effect and reveal that existing data pruning algorithms can produce highly biased classifiers. We present theoretical analysis of the classification risk "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.05579","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.05579/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.05579","created_at":"2026-07-05T10:11:18.373856+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.05579v4","created_at":"2026-07-05T10:11:18.373856+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.05579","created_at":"2026-07-05T10:11:18.373856+00:00"},{"alias_kind":"pith_short_12","alias_value":"XV7YNAZNIVKD","created_at":"2026-07-05T10:11:18.373856+00:00"},{"alias_kind":"pith_short_16","alias_value":"XV7YNAZNIVKDHJ26","created_at":"2026-07-05T10:11:18.373856+00:00"},{"alias_kind":"pith_short_8","alias_value":"XV7YNAZN","created_at":"2026-07-05T10:11:18.373856+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11761","citing_title":"RCAP: Robust, Class-Aware, Probabilistic Dynamic Dataset Pruning","ref_index":29,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XV7YNAZNIVKDHJ26BCGXZUQROO","json":"https://pith.science/pith/XV7YNAZNIVKDHJ26BCGXZUQROO.json","graph_json":"https://pith.science/api/pith-number/XV7YNAZNIVKDHJ26BCGXZUQROO/graph.json","events_json":"https://pith.science/api/pith-number/XV7YNAZNIVKDHJ26BCGXZUQROO/events.json","paper":"https://pith.science/paper/XV7YNAZN"},"agent_actions":{"view_html":"https://pith.science/pith/XV7YNAZNIVKDHJ26BCGXZUQROO","download_json":"https://pith.science/pith/XV7YNAZNIVKDHJ26BCGXZUQROO.json","view_paper":"https://pith.science/paper/XV7YNAZN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.05579&json=true","fetch_graph":"https://pith.science/api/pith-number/XV7YNAZNIVKDHJ26BCGXZUQROO/graph.json","fetch_events":"https://pith.science/api/pith-number/XV7YNAZNIVKDHJ26BCGXZUQROO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XV7YNAZNIVKDHJ26BCGXZUQROO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XV7YNAZNIVKDHJ26BCGXZUQROO/action/storage_attestation","attest_author":"https://pith.science/pith/XV7YNAZNIVKDHJ26BCGXZUQROO/action/author_attestation","sign_citation":"https://pith.science/pith/XV7YNAZNIVKDHJ26BCGXZUQROO/action/citation_signature","submit_replication":"https://pith.science/pith/XV7YNAZNIVKDHJ26BCGXZUQROO/action/replication_record"}},"created_at":"2026-07-05T10:11:18.373856+00:00","updated_at":"2026-07-05T10:11:18.373856+00:00"}