{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:VL73RD4AMNPZQMHDZTPOWZHXQT","short_pith_number":"pith:VL73RD4A","schema_version":"1.0","canonical_sha256":"aaffb88f80635f9830e3ccdeeb64f784cdb0fd39122736134b34e0f88fc3f127","source":{"kind":"arxiv","id":"2405.14908","version":4},"attestation_state":"computed","paper":{"title":"BiMix: A Bivariate Data Mixing Law for Language Model Pretraining","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Bolin Ding, Ce Ge, Daoyuan Chen, Yaliang Li, Zhijian Ma","submitted_at":"2024-05-23T09:44:02Z","abstract_excerpt":"Large language models have demonstrated remarkable capabilities across various tasks, primarily attributed to the utilization of diversely sourced data. However, the impact of pretraining data composition on model performance remains poorly understood. This paper introduces $\\textbf{BiMix}$, a novel bivariate data mixing law that models the joint scaling behavior of domain proportions and data volume in LLM pretraining. $\\textbf{BiMix}$ provides a systematic framework for understanding and optimizing data mixtures across diverse domains. Through extensive experiments on two large-scale dataset"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.14908","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-23T09:44:02Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"4e7d3f014467cdac7d07aa20d6c3c9fdc182d37bc04174203824edf3a41bf684","abstract_canon_sha256":"d5d1cab7aefb5d538935f8f207313d90790ed08ca8bc05415b4332b99238a6e4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:05:42.980687Z","signature_b64":"N9SAuPm6zsaPTovmTh2gj6OJN/FgIU4Ecx1J8k5QwS7XwtMjOJOplneQuMYzg3iipN24gZMUrlpffRPnLEG+CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"aaffb88f80635f9830e3ccdeeb64f784cdb0fd39122736134b34e0f88fc3f127","last_reissued_at":"2026-07-05T10:05:42.980201Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:05:42.980201Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"BiMix: A Bivariate Data Mixing Law for Language Model Pretraining","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Bolin Ding, Ce Ge, Daoyuan Chen, Yaliang Li, Zhijian Ma","submitted_at":"2024-05-23T09:44:02Z","abstract_excerpt":"Large language models have demonstrated remarkable capabilities across various tasks, primarily attributed to the utilization of diversely sourced data. However, the impact of pretraining data composition on model performance remains poorly understood. This paper introduces $\\textbf{BiMix}$, a novel bivariate data mixing law that models the joint scaling behavior of domain proportions and data volume in LLM pretraining. $\\textbf{BiMix}$ provides a systematic framework for understanding and optimizing data mixtures across diverse domains. Through extensive experiments on two large-scale dataset"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.14908","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.14908/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.14908","created_at":"2026-07-05T10:05:42.980261+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.14908v4","created_at":"2026-07-05T10:05:42.980261+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.14908","created_at":"2026-07-05T10:05:42.980261+00:00"},{"alias_kind":"pith_short_12","alias_value":"VL73RD4AMNPZ","created_at":"2026-07-05T10:05:42.980261+00:00"},{"alias_kind":"pith_short_16","alias_value":"VL73RD4AMNPZQMHD","created_at":"2026-07-05T10:05:42.980261+00:00"},{"alias_kind":"pith_short_8","alias_value":"VL73RD4A","created_at":"2026-07-05T10:05:42.980261+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2502.00270","citing_title":"DUET: Optimizing Training Data Mixtures via Feedback from Unseen Evaluation Tasks","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2406.11794","citing_title":"DataComp-LM: In search of the next generation of training sets for language models","ref_index":70,"is_internal_anchor":false},{"citing_arxiv_id":"2603.08022","citing_title":"Capacity-Aware Mixture Law Enables Efficient LLM Data Optimization","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16380","citing_title":"Data Mixing for Large Language Models Pretraining: A Survey and Outlook","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06859","citing_title":"Knowledge Transfer Scaling Laws for 3D Medical Imaging","ref_index":11,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VL73RD4AMNPZQMHDZTPOWZHXQT","json":"https://pith.science/pith/VL73RD4AMNPZQMHDZTPOWZHXQT.json","graph_json":"https://pith.science/api/pith-number/VL73RD4AMNPZQMHDZTPOWZHXQT/graph.json","events_json":"https://pith.science/api/pith-number/VL73RD4AMNPZQMHDZTPOWZHXQT/events.json","paper":"https://pith.science/paper/VL73RD4A"},"agent_actions":{"view_html":"https://pith.science/pith/VL73RD4AMNPZQMHDZTPOWZHXQT","download_json":"https://pith.science/pith/VL73RD4AMNPZQMHDZTPOWZHXQT.json","view_paper":"https://pith.science/paper/VL73RD4A","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.14908&json=true","fetch_graph":"https://pith.science/api/pith-number/VL73RD4AMNPZQMHDZTPOWZHXQT/graph.json","fetch_events":"https://pith.science/api/pith-number/VL73RD4AMNPZQMHDZTPOWZHXQT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VL73RD4AMNPZQMHDZTPOWZHXQT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VL73RD4AMNPZQMHDZTPOWZHXQT/action/storage_attestation","attest_author":"https://pith.science/pith/VL73RD4AMNPZQMHDZTPOWZHXQT/action/author_attestation","sign_citation":"https://pith.science/pith/VL73RD4AMNPZQMHDZTPOWZHXQT/action/citation_signature","submit_replication":"https://pith.science/pith/VL73RD4AMNPZQMHDZTPOWZHXQT/action/replication_record"}},"created_at":"2026-07-05T10:05:42.980261+00:00","updated_at":"2026-07-05T10:05:42.980261+00:00"}