{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:TTHIC7VIPFISZGDDX7GEV5LVA4","short_pith_number":"pith:TTHIC7VI","schema_version":"1.0","canonical_sha256":"9cce817ea879512c9863bfcc4af5750720fb8a24133358524fdc9a2991b3bf79","source":{"kind":"arxiv","id":"2401.12926","version":1},"attestation_state":"computed","paper":{"title":"DsDm: Model-Aware Dataset Selection with Datamodels","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Aleksander Madry, Axel Feldmann, Logan Engstrom","submitted_at":"2024-01-23T17:22:00Z","abstract_excerpt":"When selecting data for training large-scale models, standard practice is to filter for examples that match human notions of data quality. Such filtering yields qualitatively clean datapoints that intuitively should improve model behavior. However, in practice the opposite can often happen: we find that selecting according to similarity with \"high quality\" data sources may not increase (and can even hurt) performance compared to randomly selecting data.\n  To develop better methods for selecting data, we start by framing dataset selection as an optimization problem that we can directly solve fo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.12926","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-01-23T17:22:00Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"90cec4c354114e457b173ef28fb5a1868c482aceb15ed78738f2d651f2401f2c","abstract_canon_sha256":"ea2f2f01d1bd6b073da7bf1089010e15ec486397716418a7eb8d61b006d340a5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:36:45.816240Z","signature_b64":"Iu4RSfYy+AxWpNkJbdSVnqwQSyuCbtQX3DnwrM9hcWyxY4t/Ad3LACCJzu5tnXA9W2kVI0mobndwyOYlAADHBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9cce817ea879512c9863bfcc4af5750720fb8a24133358524fdc9a2991b3bf79","last_reissued_at":"2026-07-05T07:36:45.815826Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:36:45.815826Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DsDm: Model-Aware Dataset Selection with Datamodels","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Aleksander Madry, Axel Feldmann, Logan Engstrom","submitted_at":"2024-01-23T17:22:00Z","abstract_excerpt":"When selecting data for training large-scale models, standard practice is to filter for examples that match human notions of data quality. Such filtering yields qualitatively clean datapoints that intuitively should improve model behavior. However, in practice the opposite can often happen: we find that selecting according to similarity with \"high quality\" data sources may not increase (and can even hurt) performance compared to randomly selecting data.\n  To develop better methods for selecting data, we start by framing dataset selection as an optimization problem that we can directly solve fo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.12926","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.12926/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.12926","created_at":"2026-07-05T07:36:45.815875+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.12926v1","created_at":"2026-07-05T07:36:45.815875+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.12926","created_at":"2026-07-05T07:36:45.815875+00:00"},{"alias_kind":"pith_short_12","alias_value":"TTHIC7VIPFIS","created_at":"2026-07-05T07:36:45.815875+00:00"},{"alias_kind":"pith_short_16","alias_value":"TTHIC7VIPFISZGDD","created_at":"2026-07-05T07:36:45.815875+00:00"},{"alias_kind":"pith_short_8","alias_value":"TTHIC7VI","created_at":"2026-07-05T07:36:45.815875+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.15691","citing_title":"SEED: Targeted Data Selection by Weighted Independent Set","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2502.10248","citing_title":"Step-Video-T2V Technical Report: The Practice, Challenges, and Future of Video Foundation Model","ref_index":284,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08519","citing_title":"Cram Less to Fit More: Training Data Pruning Improves Memorization of Facts","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07769","citing_title":"An Empirical Study on Influence-Based Pretraining Data Selection for Code Large Language Models","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TTHIC7VIPFISZGDDX7GEV5LVA4","json":"https://pith.science/pith/TTHIC7VIPFISZGDDX7GEV5LVA4.json","graph_json":"https://pith.science/api/pith-number/TTHIC7VIPFISZGDDX7GEV5LVA4/graph.json","events_json":"https://pith.science/api/pith-number/TTHIC7VIPFISZGDDX7GEV5LVA4/events.json","paper":"https://pith.science/paper/TTHIC7VI"},"agent_actions":{"view_html":"https://pith.science/pith/TTHIC7VIPFISZGDDX7GEV5LVA4","download_json":"https://pith.science/pith/TTHIC7VIPFISZGDDX7GEV5LVA4.json","view_paper":"https://pith.science/paper/TTHIC7VI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.12926&json=true","fetch_graph":"https://pith.science/api/pith-number/TTHIC7VIPFISZGDDX7GEV5LVA4/graph.json","fetch_events":"https://pith.science/api/pith-number/TTHIC7VIPFISZGDDX7GEV5LVA4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TTHIC7VIPFISZGDDX7GEV5LVA4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TTHIC7VIPFISZGDDX7GEV5LVA4/action/storage_attestation","attest_author":"https://pith.science/pith/TTHIC7VIPFISZGDDX7GEV5LVA4/action/author_attestation","sign_citation":"https://pith.science/pith/TTHIC7VIPFISZGDDX7GEV5LVA4/action/citation_signature","submit_replication":"https://pith.science/pith/TTHIC7VIPFISZGDDX7GEV5LVA4/action/replication_record"}},"created_at":"2026-07-05T07:36:45.815875+00:00","updated_at":"2026-07-05T07:36:45.815875+00:00"}