{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:LV5LWJPVGZKNIF22DOZHX6MFZ6","short_pith_number":"pith:LV5LWJPV","schema_version":"1.0","canonical_sha256":"5d7abb25f53654d4175a1bb27bf985cfbd731b0d523f32ca42b4e5c96926153c","source":{"kind":"arxiv","id":"2403.07854","version":2},"attestation_state":"computed","paper":{"title":"Distilling the Knowledge in Data Pruning","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Adam Botach, Emanuel Ben-Baruch, G\\'erard Medioni, Igor Kviatkovsky, Manoj Aggarwal","submitted_at":"2024-03-12T17:44:45Z","abstract_excerpt":"With the increasing size of datasets used for training neural networks, data pruning becomes an attractive field of research. However, most current data pruning algorithms are limited in their ability to preserve accuracy compared to models trained on the full data, especially in high pruning regimes. In this paper we explore the application of data pruning while incorporating knowledge distillation (KD) when training on a pruned subset. That is, rather than relying solely on ground-truth labels, we also use the soft predictions from a teacher network pre-trained on the complete data. By integ"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.07854","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CV","submitted_at":"2024-03-12T17:44:45Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"3037cbfdaab8a141296f2b3c075979644cc273b407ef14958a15cb74f0a76b7c","abstract_canon_sha256":"3dae5286c1989ba58819c473cbb90b948b3a2b1bb8fa791e8376df906334a40c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:55:22.101272Z","signature_b64":"hHZokzMJSAzUQD0WtBn8Pav6NqdW5P7vJ7FiT6KboCYIJWn+hW2hDpOwTRHPuD5cwfRk/zNRBIc+a8Ir8gd6DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5d7abb25f53654d4175a1bb27bf985cfbd731b0d523f32ca42b4e5c96926153c","last_reissued_at":"2026-07-05T08:55:22.100854Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:55:22.100854Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Distilling the Knowledge in Data Pruning","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Adam Botach, Emanuel Ben-Baruch, G\\'erard Medioni, Igor Kviatkovsky, Manoj Aggarwal","submitted_at":"2024-03-12T17:44:45Z","abstract_excerpt":"With the increasing size of datasets used for training neural networks, data pruning becomes an attractive field of research. However, most current data pruning algorithms are limited in their ability to preserve accuracy compared to models trained on the full data, especially in high pruning regimes. In this paper we explore the application of data pruning while incorporating knowledge distillation (KD) when training on a pruned subset. That is, rather than relying solely on ground-truth labels, we also use the soft predictions from a teacher network pre-trained on the complete data. By integ"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.07854","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.07854/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.07854","created_at":"2026-07-05T08:55:22.100915+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.07854v2","created_at":"2026-07-05T08:55:22.100915+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.07854","created_at":"2026-07-05T08:55:22.100915+00:00"},{"alias_kind":"pith_short_12","alias_value":"LV5LWJPVGZKN","created_at":"2026-07-05T08:55:22.100915+00:00"},{"alias_kind":"pith_short_16","alias_value":"LV5LWJPVGZKNIF22","created_at":"2026-07-05T08:55:22.100915+00:00"},{"alias_kind":"pith_short_8","alias_value":"LV5LWJPV","created_at":"2026-07-05T08:55:22.100915+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.28868","citing_title":"TaxDistill: Improving Metagenomic Taxonomic Annotation via Distilled Genomic Foundation Models","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18135","citing_title":"Soft Label Pruning and Quantization for Large-Scale Dataset Distillation","ref_index":52,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LV5LWJPVGZKNIF22DOZHX6MFZ6","json":"https://pith.science/pith/LV5LWJPVGZKNIF22DOZHX6MFZ6.json","graph_json":"https://pith.science/api/pith-number/LV5LWJPVGZKNIF22DOZHX6MFZ6/graph.json","events_json":"https://pith.science/api/pith-number/LV5LWJPVGZKNIF22DOZHX6MFZ6/events.json","paper":"https://pith.science/paper/LV5LWJPV"},"agent_actions":{"view_html":"https://pith.science/pith/LV5LWJPVGZKNIF22DOZHX6MFZ6","download_json":"https://pith.science/pith/LV5LWJPVGZKNIF22DOZHX6MFZ6.json","view_paper":"https://pith.science/paper/LV5LWJPV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.07854&json=true","fetch_graph":"https://pith.science/api/pith-number/LV5LWJPVGZKNIF22DOZHX6MFZ6/graph.json","fetch_events":"https://pith.science/api/pith-number/LV5LWJPVGZKNIF22DOZHX6MFZ6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LV5LWJPVGZKNIF22DOZHX6MFZ6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LV5LWJPVGZKNIF22DOZHX6MFZ6/action/storage_attestation","attest_author":"https://pith.science/pith/LV5LWJPVGZKNIF22DOZHX6MFZ6/action/author_attestation","sign_citation":"https://pith.science/pith/LV5LWJPVGZKNIF22DOZHX6MFZ6/action/citation_signature","submit_replication":"https://pith.science/pith/LV5LWJPVGZKNIF22DOZHX6MFZ6/action/replication_record"}},"created_at":"2026-07-05T08:55:22.100915+00:00","updated_at":"2026-07-05T08:55:22.100915+00:00"}