{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:OF44L4LN6B35I7SRNJ3OAVFKG3","short_pith_number":"pith:OF44L4LN","schema_version":"1.0","canonical_sha256":"7179c5f16df077d47e516a76e054aa36f1e504ce948585e43a83d599528f17fe","source":{"kind":"arxiv","id":"2305.13169","version":2},"attestation_state":"computed","paper":{"title":"A Pretrainer's Guide to Training Data: Measuring the Effects of Data Age, Domain Coverage, Quality, & Toxicity","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Adam Roberts, Barret Zoph, Daphne Ippolito, David Mimno, Denny Zhou, Emily Reif, Gregory Yauney, Jason Wei, Katherine Lee, Kevin Robinson, Shayne Longpre","submitted_at":"2023-05-22T15:57:53Z","abstract_excerpt":"Pretraining is the preliminary and fundamental step in developing capable language models (LM). Despite this, pretraining data design is critically under-documented and often guided by empirically unsupported intuitions. To address this, we pretrain 28 1.5B parameter decoder-only models, training on data curated (1) at different times, (2) with varying toxicity and quality filters, and (3) with different domain compositions. First, we quantify the effect of pretraining data age. A temporal shift between evaluation data and pretraining data leads to performance degradation, which is not overcom"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.13169","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-05-22T15:57:53Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"9b98041546963af5c71d879c1f0709bfcd4e6aeb6343e163be2e2de57acd83bc","abstract_canon_sha256":"3702ad2f24882dc1889761569c6bee0b2c31b21b1e1470cd263ee654ef936371"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:11:55.968487Z","signature_b64":"yfelYIIn95IqQxu1WnKUXxtTtXNpYWVZiX17y76Tq5wP64XPg1pxokjf2yVANyR3LsHiJBLxbeiOUUbP5igIAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7179c5f16df077d47e516a76e054aa36f1e504ce948585e43a83d599528f17fe","last_reissued_at":"2026-07-05T07:11:55.968001Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:11:55.968001Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Pretrainer's Guide to Training Data: Measuring the Effects of Data Age, Domain Coverage, Quality, & Toxicity","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Adam Roberts, Barret Zoph, Daphne Ippolito, David Mimno, Denny Zhou, Emily Reif, Gregory Yauney, Jason Wei, Katherine Lee, Kevin Robinson, Shayne Longpre","submitted_at":"2023-05-22T15:57:53Z","abstract_excerpt":"Pretraining is the preliminary and fundamental step in developing capable language models (LM). Despite this, pretraining data design is critically under-documented and often guided by empirically unsupported intuitions. To address this, we pretrain 28 1.5B parameter decoder-only models, training on data curated (1) at different times, (2) with varying toxicity and quality filters, and (3) with different domain compositions. First, we quantify the effect of pretraining data age. A temporal shift between evaluation data and pretraining data leads to performance degradation, which is not overcom"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.13169","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.13169/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.13169","created_at":"2026-07-05T07:11:55.968074+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.13169v2","created_at":"2026-07-05T07:11:55.968074+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.13169","created_at":"2026-07-05T07:11:55.968074+00:00"},{"alias_kind":"pith_short_12","alias_value":"OF44L4LN6B35","created_at":"2026-07-05T07:11:55.968074+00:00"},{"alias_kind":"pith_short_16","alias_value":"OF44L4LN6B35I7SR","created_at":"2026-07-05T07:11:55.968074+00:00"},{"alias_kind":"pith_short_8","alias_value":"OF44L4LN","created_at":"2026-07-05T07:11:55.968074+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2305.09620","citing_title":"AI-Augmented Surveys: Leveraging Large Language Models and Surveys for Opinion Prediction","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2406.01253","citing_title":"animal2vec and MeerKAT: A self-supervised transformer for rare-event raw audio input and a large-scale reference dataset for bioacoustics","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22005","citing_title":"Check Your LLM's Secret Dictionary! Five Lines of Code Reveal What Your LLM Learned (Including What It Shouldn't Have)","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2502.10248","citing_title":"Step-Video-T2V Technical Report: The Practice, Challenges, and Future of Video Foundation Model","ref_index":227,"is_internal_anchor":false},{"citing_arxiv_id":"2507.00432","citing_title":"Does Math Reasoning Improve General LLM Capabilities? Understanding Transferability of LLM Reasoning","ref_index":238,"is_internal_anchor":false},{"citing_arxiv_id":"2406.11794","citing_title":"DataComp-LM: In search of the next generation of training sets for language models","ref_index":115,"is_internal_anchor":false},{"citing_arxiv_id":"2405.14782","citing_title":"Lessons from the Trenches on Reproducible Evaluation of Language Models","ref_index":230,"is_internal_anchor":false},{"citing_arxiv_id":"2306.11644","citing_title":"Textbooks Are All You Need","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2401.02954","citing_title":"DeepSeek LLM: Scaling Open-Source Language Models with Longtermism","ref_index":155,"is_internal_anchor":false},{"citing_arxiv_id":"2405.04434","citing_title":"DeepSeek-V2: A Strong, Economical, and Efficient Mixture-of-Experts Language Model","ref_index":153,"is_internal_anchor":false},{"citing_arxiv_id":"2303.18223","citing_title":"A Survey of Large Language Models","ref_index":231,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OF44L4LN6B35I7SRNJ3OAVFKG3","json":"https://pith.science/pith/OF44L4LN6B35I7SRNJ3OAVFKG3.json","graph_json":"https://pith.science/api/pith-number/OF44L4LN6B35I7SRNJ3OAVFKG3/graph.json","events_json":"https://pith.science/api/pith-number/OF44L4LN6B35I7SRNJ3OAVFKG3/events.json","paper":"https://pith.science/paper/OF44L4LN"},"agent_actions":{"view_html":"https://pith.science/pith/OF44L4LN6B35I7SRNJ3OAVFKG3","download_json":"https://pith.science/pith/OF44L4LN6B35I7SRNJ3OAVFKG3.json","view_paper":"https://pith.science/paper/OF44L4LN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.13169&json=true","fetch_graph":"https://pith.science/api/pith-number/OF44L4LN6B35I7SRNJ3OAVFKG3/graph.json","fetch_events":"https://pith.science/api/pith-number/OF44L4LN6B35I7SRNJ3OAVFKG3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OF44L4LN6B35I7SRNJ3OAVFKG3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OF44L4LN6B35I7SRNJ3OAVFKG3/action/storage_attestation","attest_author":"https://pith.science/pith/OF44L4LN6B35I7SRNJ3OAVFKG3/action/author_attestation","sign_citation":"https://pith.science/pith/OF44L4LN6B35I7SRNJ3OAVFKG3/action/citation_signature","submit_replication":"https://pith.science/pith/OF44L4LN6B35I7SRNJ3OAVFKG3/action/replication_record"}},"created_at":"2026-07-05T07:11:55.968074+00:00","updated_at":"2026-07-05T07:11:55.968074+00:00"}