{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:P3CCYFWMQIVEHVRQAAJRBVHH3X","short_pith_number":"pith:P3CCYFWM","schema_version":"1.0","canonical_sha256":"7ec42c16cc822a43d630001310d4e7ddf91989b9fdfd90d2164258ccc27bcd80","source":{"kind":"arxiv","id":"2309.04564","version":1},"attestation_state":"computed","paper":{"title":"When Less is More: Investigating Data Pruning for Pretraining LLMs at Scale","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Ahmet \\\"Ust\\\"un, Alex Wang, Luiza Pozzobon, Marzieh Fadaee, Max Marion, Sara Hooker","submitted_at":"2023-09-08T19:34:05Z","abstract_excerpt":"Large volumes of text data have contributed significantly to the development of large language models (LLMs) in recent years. This data is typically acquired by scraping the internet, leading to pretraining datasets comprised of noisy web text. To date, efforts to prune these datasets down to a higher quality subset have relied on hand-crafted heuristics encoded as rule-based filters. In this work, we take a wider view and explore scalable estimates of data quality that can be used to systematically measure the quality of pretraining data. We perform a rigorous comparison at scale of the simpl"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.04564","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-09-08T19:34:05Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"a1464dd88ef7a77013e5734d39d6164651dd3f1e5060b8c2a6d22987e5a50e12","abstract_canon_sha256":"1c8e9f1aabbc577b0f5b3cb24784bd12c2ec47e375ff42d228aaacc2c573be75"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:49:04.973129Z","signature_b64":"Uo9o9/WHLb63NznUVCCqIsRW2US5/Lh5Mgd1gBk3VrfWNlQOqYCHCG+w6px+MPGOoiqUK25q+BtERpSUjPDFCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7ec42c16cc822a43d630001310d4e7ddf91989b9fdfd90d2164258ccc27bcd80","last_reissued_at":"2026-07-05T06:49:04.972662Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:49:04.972662Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"When Less is More: Investigating Data Pruning for Pretraining LLMs at Scale","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Ahmet \\\"Ust\\\"un, Alex Wang, Luiza Pozzobon, Marzieh Fadaee, Max Marion, Sara Hooker","submitted_at":"2023-09-08T19:34:05Z","abstract_excerpt":"Large volumes of text data have contributed significantly to the development of large language models (LLMs) in recent years. This data is typically acquired by scraping the internet, leading to pretraining datasets comprised of noisy web text. To date, efforts to prune these datasets down to a higher quality subset have relied on hand-crafted heuristics encoded as rule-based filters. In this work, we take a wider view and explore scalable estimates of data quality that can be used to systematically measure the quality of pretraining data. We perform a rigorous comparison at scale of the simpl"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.04564","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.04564/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.04564","created_at":"2026-07-05T06:49:04.972719+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.04564v1","created_at":"2026-07-05T06:49:04.972719+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.04564","created_at":"2026-07-05T06:49:04.972719+00:00"},{"alias_kind":"pith_short_12","alias_value":"P3CCYFWMQIVE","created_at":"2026-07-05T06:49:04.972719+00:00"},{"alias_kind":"pith_short_16","alias_value":"P3CCYFWMQIVEHVRQ","created_at":"2026-07-05T06:49:04.972719+00:00"},{"alias_kind":"pith_short_8","alias_value":"P3CCYFWM","created_at":"2026-07-05T06:49:04.972719+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":20,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26091","citing_title":"On-Policy Self-Distillation with Sampled Demonstrations Reduces Output Diversity","ref_index":216,"is_internal_anchor":false},{"citing_arxiv_id":"2606.24998","citing_title":"Internal Data Repetition Destroys Language Models","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23611","citing_title":"Data Selection Through Iterative Self-Filtering for Vision-Language Settings","ref_index":195,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08574","citing_title":"OrderDP: A Theoretically Guaranteed Lossless Dynamic Data Pruning Framework","ref_index":111,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28631","citing_title":"Single-Rollout Hidden-State Dynamics for Training-Free RLVR Data Selection","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2411.05527","citing_title":"How Good is Your Wikipedia? Auditing Data Quality for Low-resource and Multilingual NLP","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2504.21850","citing_title":"Visual Compositional Tuning","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22389","citing_title":"Unified Data Selection for LLM Reasoning","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2602.18584","citing_title":"GIST: Targeted Data Selection for Instruction Tuning via Coupled Optimization Geometry","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19762","citing_title":"What Really Improves Mathematical Reasoning: Structured Reasoning Signals Beyond Pure Code","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02345","citing_title":"UI-Oceanus: Scaling GUI Agents with Synthetic Environmental Dynamics","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12906","citing_title":"Data Difficulty and the Generalization--Extrapolation Tradeoff in LLM Fine-Tuning","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2309.12284","citing_title":"MetaMath: Bootstrap Your Own Mathematical Questions for Large Language Models","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00369","citing_title":"InvEvolve: Evolving White-Box Inventory Policies via Large Language Models with Performance Guarantees","ref_index":159,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00369","citing_title":"InvEvolve: Evolving White-Box Inventory Policies via Large Language Models with Performance Guarantees","ref_index":159,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07940","citing_title":"A Systematic Framework for Tabular Data Disentanglement","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2303.18223","citing_title":"A Survey of Large Language Models","ref_index":236,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11810","citing_title":"GRACE: A Dynamic Coreset Selection Framework for Large Language Model Optimization","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05227","citing_title":"Rethinking Data Curation in LLM Training: Online Reweighting Offers Better Generalization than Offline Methods","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17396","citing_title":"Representation-Guided Parameter-Efficient LLM Unlearning","ref_index":135,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/P3CCYFWMQIVEHVRQAAJRBVHH3X","json":"https://pith.science/pith/P3CCYFWMQIVEHVRQAAJRBVHH3X.json","graph_json":"https://pith.science/api/pith-number/P3CCYFWMQIVEHVRQAAJRBVHH3X/graph.json","events_json":"https://pith.science/api/pith-number/P3CCYFWMQIVEHVRQAAJRBVHH3X/events.json","paper":"https://pith.science/paper/P3CCYFWM"},"agent_actions":{"view_html":"https://pith.science/pith/P3CCYFWMQIVEHVRQAAJRBVHH3X","download_json":"https://pith.science/pith/P3CCYFWMQIVEHVRQAAJRBVHH3X.json","view_paper":"https://pith.science/paper/P3CCYFWM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.04564&json=true","fetch_graph":"https://pith.science/api/pith-number/P3CCYFWMQIVEHVRQAAJRBVHH3X/graph.json","fetch_events":"https://pith.science/api/pith-number/P3CCYFWMQIVEHVRQAAJRBVHH3X/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/P3CCYFWMQIVEHVRQAAJRBVHH3X/action/timestamp_anchor","attest_storage":"https://pith.science/pith/P3CCYFWMQIVEHVRQAAJRBVHH3X/action/storage_attestation","attest_author":"https://pith.science/pith/P3CCYFWMQIVEHVRQAAJRBVHH3X/action/author_attestation","sign_citation":"https://pith.science/pith/P3CCYFWMQIVEHVRQAAJRBVHH3X/action/citation_signature","submit_replication":"https://pith.science/pith/P3CCYFWMQIVEHVRQAAJRBVHH3X/action/replication_record"}},"created_at":"2026-07-05T06:49:04.972719+00:00","updated_at":"2026-07-05T06:49:04.972719+00:00"}