{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:3UTT7Y2Z6HK4LXC32MPZFS6NYJ","short_pith_number":"pith:3UTT7Y2Z","schema_version":"1.0","canonical_sha256":"dd273fe359f1d5c5dc5bd31f92cbcdc26bff0d4a524aa59a2af8b6fdbf3bc14d","source":{"kind":"arxiv","id":"2505.12082","version":3},"attestation_state":"computed","paper":{"title":"Model Merging in Pre-training of Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Bairen Yi, Bole Ma, Chaoyi Zhang, Deyi Liu, Hongbin Ren, Jianqiao Lu, Jing Liu, Jin Ma, Liang Xiang, Lingjun Liu, Mengzhao Chen, Mingji Han, Minrui Wang, Shen Yan, Shiyi Zhan, Siyuan Qiao, Wenhao Hao, Xiaoying Jia, Xingyan Bin, Xunhao Lai, Xun Zhou, Yao Luo, Yiyuan Ma, Yonghui Wu, Yunshui Li, Ziwen Xu","submitted_at":"2025-05-17T16:53:14Z","abstract_excerpt":"Model merging has emerged as a promising technique for enhancing large language models, though its application in large-scale pre-training remains relatively unexplored. In this paper, we present a comprehensive investigation of model merging techniques during the pre-training process. Through extensive experiments with both dense and Mixture-of-Experts (MoE) architectures ranging from millions to over 100 billion parameters, we demonstrate that merging checkpoints trained with constant learning rates not only achieves significant performance improvements but also enables accurate prediction o"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.12082","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-17T16:53:14Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"63388d0a4dd89ea92bf8433fef2cde863288de5d0c072d6b748c328ca12d6758","abstract_canon_sha256":"a12ee78dae0b95e20003a5d2e433bf8fdfd0e96ba4e893ea41c7920d579fc3ec"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:07:24.912853Z","signature_b64":"+FyEoYQtJhPymZOTBUgzC0CYqfOcD2yETCWMbvpOR5sgfAZ2PmwZBU3fk4bbL9hTXdifr574wqrvaXQdpUTSBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dd273fe359f1d5c5dc5bd31f92cbcdc26bff0d4a524aa59a2af8b6fdbf3bc14d","last_reissued_at":"2026-07-05T11:07:24.912356Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:07:24.912356Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Model Merging in Pre-training of Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Bairen Yi, Bole Ma, Chaoyi Zhang, Deyi Liu, Hongbin Ren, Jianqiao Lu, Jing Liu, Jin Ma, Liang Xiang, Lingjun Liu, Mengzhao Chen, Mingji Han, Minrui Wang, Shen Yan, Shiyi Zhan, Siyuan Qiao, Wenhao Hao, Xiaoying Jia, Xingyan Bin, Xunhao Lai, Xun Zhou, Yao Luo, Yiyuan Ma, Yonghui Wu, Yunshui Li, Ziwen Xu","submitted_at":"2025-05-17T16:53:14Z","abstract_excerpt":"Model merging has emerged as a promising technique for enhancing large language models, though its application in large-scale pre-training remains relatively unexplored. In this paper, we present a comprehensive investigation of model merging techniques during the pre-training process. Through extensive experiments with both dense and Mixture-of-Experts (MoE) architectures ranging from millions to over 100 billion parameters, we demonstrate that merging checkpoints trained with constant learning rates not only achieves significant performance improvements but also enables accurate prediction o"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.12082","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.12082/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.12082","created_at":"2026-07-05T11:07:24.912413+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.12082v3","created_at":"2026-07-05T11:07:24.912413+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.12082","created_at":"2026-07-05T11:07:24.912413+00:00"},{"alias_kind":"pith_short_12","alias_value":"3UTT7Y2Z6HK4","created_at":"2026-07-05T11:07:24.912413+00:00"},{"alias_kind":"pith_short_16","alias_value":"3UTT7Y2Z6HK4LXC3","created_at":"2026-07-05T11:07:24.912413+00:00"},{"alias_kind":"pith_short_8","alias_value":"3UTT7Y2Z","created_at":"2026-07-05T11:07:24.912413+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22971","citing_title":"Humanoid-OmniOcc: Stereo-Based Full-View Occupancy Dataset for Embodied AI","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2606.24901","citing_title":"LLM Evolution as an Industry-Scale Ecosystem: A Lifecycle Perspective on Continual Learning","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02291","citing_title":"Optimizing Visual Generative Models via Distribution-wise Rewards","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10651","citing_title":"Kwai Keye-VL-2.0 Technical Report","ref_index":89,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23061","citing_title":"Anytime Training with Schedule-Free Spectral Optimization","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19095","citing_title":"ScheduleFree+: Scaling Learning-Rate-Free & Schedule-Free Learning to Large Language Models","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2408.07666","citing_title":"Model Merging in LLMs, MLLMs, and Beyond: Methods, Theories, Applications and Opportunities","ref_index":129,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3UTT7Y2Z6HK4LXC32MPZFS6NYJ","json":"https://pith.science/pith/3UTT7Y2Z6HK4LXC32MPZFS6NYJ.json","graph_json":"https://pith.science/api/pith-number/3UTT7Y2Z6HK4LXC32MPZFS6NYJ/graph.json","events_json":"https://pith.science/api/pith-number/3UTT7Y2Z6HK4LXC32MPZFS6NYJ/events.json","paper":"https://pith.science/paper/3UTT7Y2Z"},"agent_actions":{"view_html":"https://pith.science/pith/3UTT7Y2Z6HK4LXC32MPZFS6NYJ","download_json":"https://pith.science/pith/3UTT7Y2Z6HK4LXC32MPZFS6NYJ.json","view_paper":"https://pith.science/paper/3UTT7Y2Z","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.12082&json=true","fetch_graph":"https://pith.science/api/pith-number/3UTT7Y2Z6HK4LXC32MPZFS6NYJ/graph.json","fetch_events":"https://pith.science/api/pith-number/3UTT7Y2Z6HK4LXC32MPZFS6NYJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3UTT7Y2Z6HK4LXC32MPZFS6NYJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3UTT7Y2Z6HK4LXC32MPZFS6NYJ/action/storage_attestation","attest_author":"https://pith.science/pith/3UTT7Y2Z6HK4LXC32MPZFS6NYJ/action/author_attestation","sign_citation":"https://pith.science/pith/3UTT7Y2Z6HK4LXC32MPZFS6NYJ/action/citation_signature","submit_replication":"https://pith.science/pith/3UTT7Y2Z6HK4LXC32MPZFS6NYJ/action/replication_record"}},"created_at":"2026-07-05T11:07:24.912413+00:00","updated_at":"2026-07-05T11:07:24.912413+00:00"}