{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:UH6EVES4MKTFFGDLSCDLIXDRGV","short_pith_number":"pith:UH6EVES4","schema_version":"1.0","canonical_sha256":"a1fc4a925c62a652986b9086b45c71355fa05767d07ac0c970ce7e866c5328d8","source":{"kind":"arxiv","id":"2409.17115","version":2},"attestation_state":"computed","paper":{"title":"Programming Every Example: Lifting Pre-training Data Quality Like Experts at Scale","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Fan Zhou, Junlong Li, Pengfei Liu, Qian Liu, Zengzhi Wang","submitted_at":"2024-09-25T17:28:13Z","abstract_excerpt":"Large language model pre-training has traditionally relied on human experts to craft heuristics for improving the corpora quality, resulting in numerous rules developed to date. However, these rules lack the flexibility to address the unique characteristics of individual example effectively. Meanwhile, applying tailored rules to every example is impractical for human experts. In this paper, we demonstrate that even small language models, with as few as 0.3B parameters, can exhibit substantial data refining capabilities comparable to those of human experts. We introduce Programming Every Exampl"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.17115","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-09-25T17:28:13Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"e304c90624d8fa5e2f939e79d343deb2fcaab825daecce8050765285acb64e5b","abstract_canon_sha256":"97339413b55ec509ad90f6ff71cdb4902157f80210240287ef220e58a508195c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:14:24.185326Z","signature_b64":"ZnPdrs0buj550wVe0x8r71c5F2TxN7yGftkeWXuaReH6aq7fh+D4xN3GGzcFDKZ/7yOHSgzLcFxnOi5FNRovBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a1fc4a925c62a652986b9086b45c71355fa05767d07ac0c970ce7e866c5328d8","last_reissued_at":"2026-07-05T10:14:24.184760Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:14:24.184760Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Programming Every Example: Lifting Pre-training Data Quality Like Experts at Scale","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Fan Zhou, Junlong Li, Pengfei Liu, Qian Liu, Zengzhi Wang","submitted_at":"2024-09-25T17:28:13Z","abstract_excerpt":"Large language model pre-training has traditionally relied on human experts to craft heuristics for improving the corpora quality, resulting in numerous rules developed to date. However, these rules lack the flexibility to address the unique characteristics of individual example effectively. Meanwhile, applying tailored rules to every example is impractical for human experts. In this paper, we demonstrate that even small language models, with as few as 0.3B parameters, can exhibit substantial data refining capabilities comparable to those of human experts. We introduce Programming Every Exampl"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.17115","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.17115/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.17115","created_at":"2026-07-05T10:14:24.184829+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.17115v2","created_at":"2026-07-05T10:14:24.184829+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.17115","created_at":"2026-07-05T10:14:24.184829+00:00"},{"alias_kind":"pith_short_12","alias_value":"UH6EVES4MKTF","created_at":"2026-07-05T10:14:24.184829+00:00"},{"alias_kind":"pith_short_16","alias_value":"UH6EVES4MKTFFGDL","created_at":"2026-07-05T10:14:24.184829+00:00"},{"alias_kind":"pith_short_8","alias_value":"UH6EVES4","created_at":"2026-07-05T10:14:24.184829+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.19762","citing_title":"What Really Improves Mathematical Reasoning: Structured Reasoning Signals Beyond Pure Code","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2510.06499","citing_title":"Webscale-RL: Automated Data Pipeline for Scaling RL Data to Pretraining Levels","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2603.18297","citing_title":"Path-Constrained Mixture-of-Experts","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2502.05171","citing_title":"Scaling up Test-Time Compute with Latent Reasoning: A Recurrent Depth Approach","ref_index":186,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UH6EVES4MKTFFGDLSCDLIXDRGV","json":"https://pith.science/pith/UH6EVES4MKTFFGDLSCDLIXDRGV.json","graph_json":"https://pith.science/api/pith-number/UH6EVES4MKTFFGDLSCDLIXDRGV/graph.json","events_json":"https://pith.science/api/pith-number/UH6EVES4MKTFFGDLSCDLIXDRGV/events.json","paper":"https://pith.science/paper/UH6EVES4"},"agent_actions":{"view_html":"https://pith.science/pith/UH6EVES4MKTFFGDLSCDLIXDRGV","download_json":"https://pith.science/pith/UH6EVES4MKTFFGDLSCDLIXDRGV.json","view_paper":"https://pith.science/paper/UH6EVES4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.17115&json=true","fetch_graph":"https://pith.science/api/pith-number/UH6EVES4MKTFFGDLSCDLIXDRGV/graph.json","fetch_events":"https://pith.science/api/pith-number/UH6EVES4MKTFFGDLSCDLIXDRGV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UH6EVES4MKTFFGDLSCDLIXDRGV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UH6EVES4MKTFFGDLSCDLIXDRGV/action/storage_attestation","attest_author":"https://pith.science/pith/UH6EVES4MKTFFGDLSCDLIXDRGV/action/author_attestation","sign_citation":"https://pith.science/pith/UH6EVES4MKTFFGDLSCDLIXDRGV/action/citation_signature","submit_replication":"https://pith.science/pith/UH6EVES4MKTFFGDLSCDLIXDRGV/action/replication_record"}},"created_at":"2026-07-05T10:14:24.184829+00:00","updated_at":"2026-07-05T10:14:24.184829+00:00"}