{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:2BX6UNZZFJHYPVHDBJBFLPOZBI","short_pith_number":"pith:2BX6UNZZ","schema_version":"1.0","canonical_sha256":"d06fea37392a4f87d4e30a4255bdd90a30eef0e130323e73cae705b4b428f7f2","source":{"kind":"arxiv","id":"2312.05599","version":1},"attestation_state":"computed","paper":{"title":"Not All Data Matters: An End-to-End Adaptive Dataset Pruning Framework for Enhancing Model Performance and Efficiency","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Furao Shen, Hongchao Yang, Jian Zhao, Suhan Guo, Suorong Yang","submitted_at":"2023-12-09T16:01:21Z","abstract_excerpt":"While deep neural networks have demonstrated remarkable performance across various tasks, they typically require massive training data. Due to the presence of redundancies and biases in real-world datasets, not all data in the training dataset contributes to the model performance. To address this issue, dataset pruning techniques have been introduced to enhance model performance and efficiency by eliminating redundant training samples and reducing computational and memory overhead. However, previous works most rely on manually crafted scalar scores, limiting their practical performance and sca"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.05599","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2023-12-09T16:01:21Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"31018ccad88c37e891001a747c0ae7e5155d3d76d2948ffc9738d9a82f1f20b5","abstract_canon_sha256":"6c5aaa5e3726ab75a9eb3b656c9ec89e96c77c4636c0ce535735101b9b0b12c1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:22:29.546254Z","signature_b64":"TuIaAd+/RYSXm2T5K9Y1TWCN5QE0LybS78708XAlmYe174BzRWtZFLOBRI0DJC7eF3CkQWqRhaBzh+zh9i2IAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d06fea37392a4f87d4e30a4255bdd90a30eef0e130323e73cae705b4b428f7f2","last_reissued_at":"2026-07-05T07:22:29.544434Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:22:29.544434Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Not All Data Matters: An End-to-End Adaptive Dataset Pruning Framework for Enhancing Model Performance and Efficiency","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Furao Shen, Hongchao Yang, Jian Zhao, Suhan Guo, Suorong Yang","submitted_at":"2023-12-09T16:01:21Z","abstract_excerpt":"While deep neural networks have demonstrated remarkable performance across various tasks, they typically require massive training data. Due to the presence of redundancies and biases in real-world datasets, not all data in the training dataset contributes to the model performance. To address this issue, dataset pruning techniques have been introduced to enhance model performance and efficiency by eliminating redundant training samples and reducing computational and memory overhead. However, previous works most rely on manually crafted scalar scores, limiting their practical performance and sca"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.05599","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.05599/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.05599","created_at":"2026-07-05T07:22:29.544487+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.05599v1","created_at":"2026-07-05T07:22:29.544487+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.05599","created_at":"2026-07-05T07:22:29.544487+00:00"},{"alias_kind":"pith_short_12","alias_value":"2BX6UNZZFJHY","created_at":"2026-07-05T07:22:29.544487+00:00"},{"alias_kind":"pith_short_16","alias_value":"2BX6UNZZFJHYPVHD","created_at":"2026-07-05T07:22:29.544487+00:00"},{"alias_kind":"pith_short_8","alias_value":"2BX6UNZZ","created_at":"2026-07-05T07:22:29.544487+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.12750","citing_title":"Multimodal-Guided Dynamic Dataset Pruning for Robust and Efficient Data-Centric Learning","ref_index":18,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2BX6UNZZFJHYPVHDBJBFLPOZBI","json":"https://pith.science/pith/2BX6UNZZFJHYPVHDBJBFLPOZBI.json","graph_json":"https://pith.science/api/pith-number/2BX6UNZZFJHYPVHDBJBFLPOZBI/graph.json","events_json":"https://pith.science/api/pith-number/2BX6UNZZFJHYPVHDBJBFLPOZBI/events.json","paper":"https://pith.science/paper/2BX6UNZZ"},"agent_actions":{"view_html":"https://pith.science/pith/2BX6UNZZFJHYPVHDBJBFLPOZBI","download_json":"https://pith.science/pith/2BX6UNZZFJHYPVHDBJBFLPOZBI.json","view_paper":"https://pith.science/paper/2BX6UNZZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.05599&json=true","fetch_graph":"https://pith.science/api/pith-number/2BX6UNZZFJHYPVHDBJBFLPOZBI/graph.json","fetch_events":"https://pith.science/api/pith-number/2BX6UNZZFJHYPVHDBJBFLPOZBI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2BX6UNZZFJHYPVHDBJBFLPOZBI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2BX6UNZZFJHYPVHDBJBFLPOZBI/action/storage_attestation","attest_author":"https://pith.science/pith/2BX6UNZZFJHYPVHDBJBFLPOZBI/action/author_attestation","sign_citation":"https://pith.science/pith/2BX6UNZZFJHYPVHDBJBFLPOZBI/action/citation_signature","submit_replication":"https://pith.science/pith/2BX6UNZZFJHYPVHDBJBFLPOZBI/action/replication_record"}},"created_at":"2026-07-05T07:22:29.544487+00:00","updated_at":"2026-07-05T07:22:29.544487+00:00"}