{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:QQIKCYCHRQ3YZJ5MNOCAUGUW5N","short_pith_number":"pith:QQIKCYCH","schema_version":"1.0","canonical_sha256":"8410a160478c378ca7ac6b840a1a96eb4355708c8dfe2b6552c8084676320119","source":{"kind":"arxiv","id":"2406.03476","version":1},"attestation_state":"computed","paper":{"title":"Does your data spark joy? Performance gains from domain upsampling at the end of training","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Brett W. Larsen, Cody Blakeney, Jonathan Frankle, Mansheej Paul, Sean Owen","submitted_at":"2024-06-05T17:29:15Z","abstract_excerpt":"Pretraining datasets for large language models (LLMs) have grown to trillions of tokens composed of large amounts of CommonCrawl (CC) web scrape along with smaller, domain-specific datasets. It is expensive to understand the impact of these domain-specific datasets on model capabilities as training at large FLOP scales is required to reveal significant changes to difficult and emergent benchmarks. Given the increasing cost of experimenting with pretraining data, how does one determine the optimal balance between the diversity in general web scrapes and the information density of domain specifi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.03476","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2024-06-05T17:29:15Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"3c22f9f3db56181a4181b192984d163d59d6bc6ff5ca2f012b0337da102694ab","abstract_canon_sha256":"8adff10cd38dfa1509e2e40fb3095d77f58c3c418a77f9c411643a797bd37c0d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:27:57.815969Z","signature_b64":"s7Abr2iEuZjVSDRbsHGltom9CJCGrz8jhp9uuBAbVlFgCBqAG+ccYKnUwBcXBauGK70v9w8Ttrp5P7+pvVWnBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8410a160478c378ca7ac6b840a1a96eb4355708c8dfe2b6552c8084676320119","last_reissued_at":"2026-07-05T08:27:57.815412Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:27:57.815412Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Does your data spark joy? Performance gains from domain upsampling at the end of training","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Brett W. Larsen, Cody Blakeney, Jonathan Frankle, Mansheej Paul, Sean Owen","submitted_at":"2024-06-05T17:29:15Z","abstract_excerpt":"Pretraining datasets for large language models (LLMs) have grown to trillions of tokens composed of large amounts of CommonCrawl (CC) web scrape along with smaller, domain-specific datasets. It is expensive to understand the impact of these domain-specific datasets on model capabilities as training at large FLOP scales is required to reveal significant changes to difficult and emergent benchmarks. Given the increasing cost of experimenting with pretraining data, how does one determine the optimal balance between the diversity in general web scrapes and the information density of domain specifi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.03476","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.03476/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.03476","created_at":"2026-07-05T08:27:57.815479+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.03476v1","created_at":"2026-07-05T08:27:57.815479+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.03476","created_at":"2026-07-05T08:27:57.815479+00:00"},{"alias_kind":"pith_short_12","alias_value":"QQIKCYCHRQ3Y","created_at":"2026-07-05T08:27:57.815479+00:00"},{"alias_kind":"pith_short_16","alias_value":"QQIKCYCHRQ3YZJ5M","created_at":"2026-07-05T08:27:57.815479+00:00"},{"alias_kind":"pith_short_8","alias_value":"QQIKCYCH","created_at":"2026-07-05T08:27:57.815479+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24320","citing_title":"ZONOS2 Technical Report","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.24320","citing_title":"ZONOS2 Technical Report","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2603.08022","citing_title":"Capacity-Aware Mixture Law Enables Efficient LLM Data Optimization","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2502.02737","citing_title":"SmolLM2: When Smol Goes Big -- Data-Centric Training of a Small Language Model","ref_index":153,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05365","citing_title":"ZAYA1-8B Technical Report","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2408.03314","citing_title":"Scaling LLM Test-Time Compute Optimally can be More Effective than Scaling Model Parameters","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00072","citing_title":"XekRung Technical Report","ref_index":131,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QQIKCYCHRQ3YZJ5MNOCAUGUW5N","json":"https://pith.science/pith/QQIKCYCHRQ3YZJ5MNOCAUGUW5N.json","graph_json":"https://pith.science/api/pith-number/QQIKCYCHRQ3YZJ5MNOCAUGUW5N/graph.json","events_json":"https://pith.science/api/pith-number/QQIKCYCHRQ3YZJ5MNOCAUGUW5N/events.json","paper":"https://pith.science/paper/QQIKCYCH"},"agent_actions":{"view_html":"https://pith.science/pith/QQIKCYCHRQ3YZJ5MNOCAUGUW5N","download_json":"https://pith.science/pith/QQIKCYCHRQ3YZJ5MNOCAUGUW5N.json","view_paper":"https://pith.science/paper/QQIKCYCH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.03476&json=true","fetch_graph":"https://pith.science/api/pith-number/QQIKCYCHRQ3YZJ5MNOCAUGUW5N/graph.json","fetch_events":"https://pith.science/api/pith-number/QQIKCYCHRQ3YZJ5MNOCAUGUW5N/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QQIKCYCHRQ3YZJ5MNOCAUGUW5N/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QQIKCYCHRQ3YZJ5MNOCAUGUW5N/action/storage_attestation","attest_author":"https://pith.science/pith/QQIKCYCHRQ3YZJ5MNOCAUGUW5N/action/author_attestation","sign_citation":"https://pith.science/pith/QQIKCYCHRQ3YZJ5MNOCAUGUW5N/action/citation_signature","submit_replication":"https://pith.science/pith/QQIKCYCHRQ3YZJ5MNOCAUGUW5N/action/replication_record"}},"created_at":"2026-07-05T08:27:57.815479+00:00","updated_at":"2026-07-05T08:27:57.815479+00:00"}