{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:KZU3EDUEQT4DMVI3MTLKSZZSMS","short_pith_number":"pith:KZU3EDUE","schema_version":"1.0","canonical_sha256":"5669b20e8484f836551b64d6a9673264ab89584f6c4e5c5f1d6408f886de1e28","source":{"kind":"arxiv","id":"2502.16802","version":3},"attestation_state":"computed","paper":{"title":"Topic Over Source: The Key to Effective Data Mixing for Language Models Pre-training","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Conghui He, He Zhu, Jiahui Peng, Jiantao Qiu, Jing Yu, Ren Ma, Xinlin Zhuang","submitted_at":"2025-02-24T03:25:56Z","abstract_excerpt":"The performance of large language models (LLMs) is significantly affected by the quality and composition of their pre-training data, which is inherently diverse, spanning various languages, sources, and topics. Effectively integrating these heterogeneous data groups is crucial for optimizing LLM performance. Previous research has predominantly concentrated on source-based data mixing, often neglecting the nuanced topic-level characteristics of the data. To address this gap, we propose a topic-based data mixing strategy that utilizes detailed topic labels generated through a multi-stage process"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.16802","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-02-24T03:25:56Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"3d8c228d5434fbd556c0056d8bf7457bd7923f893b7fafa998a0be0e063466e7","abstract_canon_sha256":"da413d1ada2601b78f8abf1b2e65d30ff929bc429ec0ea51c899e2da09159646"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:50:31.684095Z","signature_b64":"IdjvVkMFnafoOcKVyrNLKefFBy7v2nuU91D3f5DYNmIDX5BFjXY9dZr0bFIrLLtXXgmiBPcyqQArkgbDL97gAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5669b20e8484f836551b64d6a9673264ab89584f6c4e5c5f1d6408f886de1e28","last_reissued_at":"2026-07-05T11:50:31.683577Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:50:31.683577Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Topic Over Source: The Key to Effective Data Mixing for Language Models Pre-training","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Conghui He, He Zhu, Jiahui Peng, Jiantao Qiu, Jing Yu, Ren Ma, Xinlin Zhuang","submitted_at":"2025-02-24T03:25:56Z","abstract_excerpt":"The performance of large language models (LLMs) is significantly affected by the quality and composition of their pre-training data, which is inherently diverse, spanning various languages, sources, and topics. Effectively integrating these heterogeneous data groups is crucial for optimizing LLM performance. Previous research has predominantly concentrated on source-based data mixing, often neglecting the nuanced topic-level characteristics of the data. To address this gap, we propose a topic-based data mixing strategy that utilizes detailed topic labels generated through a multi-stage process"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.16802","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.16802/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.16802","created_at":"2026-07-05T11:50:31.683644+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.16802v3","created_at":"2026-07-05T11:50:31.683644+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.16802","created_at":"2026-07-05T11:50:31.683644+00:00"},{"alias_kind":"pith_short_12","alias_value":"KZU3EDUEQT4D","created_at":"2026-07-05T11:50:31.683644+00:00"},{"alias_kind":"pith_short_16","alias_value":"KZU3EDUEQT4DMVI3","created_at":"2026-07-05T11:50:31.683644+00:00"},{"alias_kind":"pith_short_8","alias_value":"KZU3EDUE","created_at":"2026-07-05T11:50:31.683644+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24133","citing_title":"Holistic Data Scheduler for LLM Pre-training via Multi-Objective Reinforcement Learning","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02266","citing_title":"HERMES: A Multi-Granularity Labeling Substrate for Pre-training Data Mixtures","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16380","citing_title":"Data Mixing for Large Language Models Pretraining: A Survey and Outlook","ref_index":61,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KZU3EDUEQT4DMVI3MTLKSZZSMS","json":"https://pith.science/pith/KZU3EDUEQT4DMVI3MTLKSZZSMS.json","graph_json":"https://pith.science/api/pith-number/KZU3EDUEQT4DMVI3MTLKSZZSMS/graph.json","events_json":"https://pith.science/api/pith-number/KZU3EDUEQT4DMVI3MTLKSZZSMS/events.json","paper":"https://pith.science/paper/KZU3EDUE"},"agent_actions":{"view_html":"https://pith.science/pith/KZU3EDUEQT4DMVI3MTLKSZZSMS","download_json":"https://pith.science/pith/KZU3EDUEQT4DMVI3MTLKSZZSMS.json","view_paper":"https://pith.science/paper/KZU3EDUE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.16802&json=true","fetch_graph":"https://pith.science/api/pith-number/KZU3EDUEQT4DMVI3MTLKSZZSMS/graph.json","fetch_events":"https://pith.science/api/pith-number/KZU3EDUEQT4DMVI3MTLKSZZSMS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KZU3EDUEQT4DMVI3MTLKSZZSMS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KZU3EDUEQT4DMVI3MTLKSZZSMS/action/storage_attestation","attest_author":"https://pith.science/pith/KZU3EDUEQT4DMVI3MTLKSZZSMS/action/author_attestation","sign_citation":"https://pith.science/pith/KZU3EDUEQT4DMVI3MTLKSZZSMS/action/citation_signature","submit_replication":"https://pith.science/pith/KZU3EDUEQT4DMVI3MTLKSZZSMS/action/replication_record"}},"created_at":"2026-07-05T11:50:31.683644+00:00","updated_at":"2026-07-05T11:50:31.683644+00:00"}