{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:3RZFV2CFJV2Q6Y6JLC67BYSC3X","short_pith_number":"pith:3RZFV2CF","schema_version":"1.0","canonical_sha256":"dc725ae8454d750f63c958bdf0e242ddf8824be8c275a78cd1fa75ac2fc44ae7","source":{"kind":"arxiv","id":"2312.02406","version":2},"attestation_state":"computed","paper":{"title":"Efficient Online Data Mixing For Language Model Pre-Training","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Alon Albalak, Colin Raffel, Liangming Pan, William Yang Wang","submitted_at":"2023-12-05T00:42:35Z","abstract_excerpt":"The data used to pretrain large language models has a decisive impact on a model's downstream performance, which has led to a large body of work on data selection methods that aim to automatically determine the most suitable data to use for pretraining. Existing data selection methods suffer from slow and computationally expensive processes, a problem amplified by the increasing size of models and of pretraining datasets. Data mixing, on the other hand, reduces the complexity of data selection by grouping data points together and determining sampling probabilities across entire groups. However"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.02406","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-12-05T00:42:35Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"6f251656da08c239524c85fa43bd949e6c323292d7bc2f8a4e2ced386251ce6a","abstract_canon_sha256":"8616cc4ed936e9d8b56240d268127027e8acfa7b9e7a569561fabe81076fcfce"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:22:01.886114Z","signature_b64":"uxCuCAIXj/xhjfXGMq3lM0qsyaz/5uhsDJBPYVfSb4ZMlsYYOtPWBMQqBwMcGhxkStxmNh6CjODnwmrPJxpOCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dc725ae8454d750f63c958bdf0e242ddf8824be8c275a78cd1fa75ac2fc44ae7","last_reissued_at":"2026-07-05T07:22:01.885623Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:22:01.885623Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Efficient Online Data Mixing For Language Model Pre-Training","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Alon Albalak, Colin Raffel, Liangming Pan, William Yang Wang","submitted_at":"2023-12-05T00:42:35Z","abstract_excerpt":"The data used to pretrain large language models has a decisive impact on a model's downstream performance, which has led to a large body of work on data selection methods that aim to automatically determine the most suitable data to use for pretraining. Existing data selection methods suffer from slow and computationally expensive processes, a problem amplified by the increasing size of models and of pretraining datasets. Data mixing, on the other hand, reduces the complexity of data selection by grouping data points together and determining sampling probabilities across entire groups. However"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.02406","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.02406/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.02406","created_at":"2026-07-05T07:22:01.885685+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.02406v2","created_at":"2026-07-05T07:22:01.885685+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.02406","created_at":"2026-07-05T07:22:01.885685+00:00"},{"alias_kind":"pith_short_12","alias_value":"3RZFV2CFJV2Q","created_at":"2026-07-05T07:22:01.885685+00:00"},{"alias_kind":"pith_short_16","alias_value":"3RZFV2CFJV2Q6Y6J","created_at":"2026-07-05T07:22:01.885685+00:00"},{"alias_kind":"pith_short_8","alias_value":"3RZFV2CF","created_at":"2026-07-05T07:22:01.885685+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24133","citing_title":"Holistic Data Scheduler for LLM Pre-training via Multi-Objective Reinforcement Learning","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18307","citing_title":"DRIFT: Refining Instruction Data via On-Policy Data Attribution","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01686","citing_title":"WARP: Weight-Space Analysis for Recovering Training Data Portfolios","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08167","citing_title":"Explaining Data Mixing Scaling Laws","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29858","citing_title":"Smooth Scaling Laws Hide Stepwise Token Learning","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07597","citing_title":"Repetition Mismatch: Why Data Mixture Experiments Don't Scale and How to Fix Them","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2502.00270","citing_title":"DUET: Optimizing Training Data Mixtures via Feedback from Unseen Evaluation Tasks","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2406.11794","citing_title":"DataComp-LM: In search of the next generation of training sets for language models","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14350","citing_title":"Distributionally Robust Multi-Task Reinforcement Learning via Adaptive Task Sampling","ref_index":268,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16380","citing_title":"Data Mixing for Large Language Models Pretraining: A Survey and Outlook","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14198","citing_title":"MixAtlas: Uncertainty-aware Data Mixture Optimization for Multimodal LLM Midtraining","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3RZFV2CFJV2Q6Y6JLC67BYSC3X","json":"https://pith.science/pith/3RZFV2CFJV2Q6Y6JLC67BYSC3X.json","graph_json":"https://pith.science/api/pith-number/3RZFV2CFJV2Q6Y6JLC67BYSC3X/graph.json","events_json":"https://pith.science/api/pith-number/3RZFV2CFJV2Q6Y6JLC67BYSC3X/events.json","paper":"https://pith.science/paper/3RZFV2CF"},"agent_actions":{"view_html":"https://pith.science/pith/3RZFV2CFJV2Q6Y6JLC67BYSC3X","download_json":"https://pith.science/pith/3RZFV2CFJV2Q6Y6JLC67BYSC3X.json","view_paper":"https://pith.science/paper/3RZFV2CF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.02406&json=true","fetch_graph":"https://pith.science/api/pith-number/3RZFV2CFJV2Q6Y6JLC67BYSC3X/graph.json","fetch_events":"https://pith.science/api/pith-number/3RZFV2CFJV2Q6Y6JLC67BYSC3X/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3RZFV2CFJV2Q6Y6JLC67BYSC3X/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3RZFV2CFJV2Q6Y6JLC67BYSC3X/action/storage_attestation","attest_author":"https://pith.science/pith/3RZFV2CFJV2Q6Y6JLC67BYSC3X/action/author_attestation","sign_citation":"https://pith.science/pith/3RZFV2CFJV2Q6Y6JLC67BYSC3X/action/citation_signature","submit_replication":"https://pith.science/pith/3RZFV2CFJV2Q6Y6JLC67BYSC3X/action/replication_record"}},"created_at":"2026-07-05T07:22:01.885685+00:00","updated_at":"2026-07-05T07:22:01.885685+00:00"}