{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:VA65YLLAYW2YLQV674X2SOPLRC","short_pith_number":"pith:VA65YLLA","schema_version":"1.0","canonical_sha256":"a83ddc2d60c5b585c2beff2fa939eb88bfce7032c47c6a9b2a0056d8a2021cd0","source":{"kind":"arxiv","id":"2408.03092","version":1},"attestation_state":"computed","paper":{"title":"Extend Model Merging from Fine-Tuned to Pre-Trained Large Language Models via Weight Disentanglement","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bowen Yu, Fei Huang, Haiyang Yu, Le Yu, Yongbin Li","submitted_at":"2024-08-06T10:46:46Z","abstract_excerpt":"Merging Large Language Models (LLMs) aims to amalgamate multiple homologous LLMs into one with all the capabilities. Ideally, any LLMs sharing the same backbone should be mergeable, irrespective of whether they are Fine-Tuned (FT) with minor parameter changes or Pre-Trained (PT) with substantial parameter shifts. However, existing methods often manually assign the model importance, rendering them feasible only for LLMs with similar parameter alterations, such as multiple FT LLMs. The diverse parameter changed ranges between FT and PT LLMs pose challenges for current solutions in empirically de"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.03092","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-08-06T10:46:46Z","cross_cats_sorted":[],"title_canon_sha256":"83354f6c6cb91f83df4f910e6ae3f5a3c3f70a3f8cd50cc28dc6fced0945b4b1","abstract_canon_sha256":"dd6181a90e3e3a650bfee62f8e83fe151dea83e9a4452bf9e9dab46fcd73fcf6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:52:45.369161Z","signature_b64":"OnJzVaPAnuMiPFd/XnkaGtyYejAyt0eS773axlrI+a1qdhpypRyqefzAsow5DSrGdQex9OLmZ2+BJP/SJP3DBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a83ddc2d60c5b585c2beff2fa939eb88bfce7032c47c6a9b2a0056d8a2021cd0","last_reissued_at":"2026-07-05T08:52:45.368678Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:52:45.368678Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Extend Model Merging from Fine-Tuned to Pre-Trained Large Language Models via Weight Disentanglement","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bowen Yu, Fei Huang, Haiyang Yu, Le Yu, Yongbin Li","submitted_at":"2024-08-06T10:46:46Z","abstract_excerpt":"Merging Large Language Models (LLMs) aims to amalgamate multiple homologous LLMs into one with all the capabilities. Ideally, any LLMs sharing the same backbone should be mergeable, irrespective of whether they are Fine-Tuned (FT) with minor parameter changes or Pre-Trained (PT) with substantial parameter shifts. However, existing methods often manually assign the model importance, rendering them feasible only for LLMs with similar parameter alterations, such as multiple FT LLMs. The diverse parameter changed ranges between FT and PT LLMs pose challenges for current solutions in empirically de"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.03092","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.03092/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.03092","created_at":"2026-07-05T08:52:45.368736+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.03092v1","created_at":"2026-07-05T08:52:45.368736+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.03092","created_at":"2026-07-05T08:52:45.368736+00:00"},{"alias_kind":"pith_short_12","alias_value":"VA65YLLAYW2Y","created_at":"2026-07-05T08:52:45.368736+00:00"},{"alias_kind":"pith_short_16","alias_value":"VA65YLLAYW2YLQV6","created_at":"2026-07-05T08:52:45.368736+00:00"},{"alias_kind":"pith_short_8","alias_value":"VA65YLLA","created_at":"2026-07-05T08:52:45.368736+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2601.13992","citing_title":"\"The Whole Is Greater Than the Sum of Its Parts\": A Compatibility-Aware Multi-Teacher CoT Distillation Framework","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2408.07666","citing_title":"Model Merging in LLMs, MLLMs, and Beyond: Methods, Theories, Applications and Opportunities","ref_index":279,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13694","citing_title":"Weight Patching: Toward Source-Level Mechanistic Localization in LLMs","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VA65YLLAYW2YLQV674X2SOPLRC","json":"https://pith.science/pith/VA65YLLAYW2YLQV674X2SOPLRC.json","graph_json":"https://pith.science/api/pith-number/VA65YLLAYW2YLQV674X2SOPLRC/graph.json","events_json":"https://pith.science/api/pith-number/VA65YLLAYW2YLQV674X2SOPLRC/events.json","paper":"https://pith.science/paper/VA65YLLA"},"agent_actions":{"view_html":"https://pith.science/pith/VA65YLLAYW2YLQV674X2SOPLRC","download_json":"https://pith.science/pith/VA65YLLAYW2YLQV674X2SOPLRC.json","view_paper":"https://pith.science/paper/VA65YLLA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.03092&json=true","fetch_graph":"https://pith.science/api/pith-number/VA65YLLAYW2YLQV674X2SOPLRC/graph.json","fetch_events":"https://pith.science/api/pith-number/VA65YLLAYW2YLQV674X2SOPLRC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VA65YLLAYW2YLQV674X2SOPLRC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VA65YLLAYW2YLQV674X2SOPLRC/action/storage_attestation","attest_author":"https://pith.science/pith/VA65YLLAYW2YLQV674X2SOPLRC/action/author_attestation","sign_citation":"https://pith.science/pith/VA65YLLAYW2YLQV674X2SOPLRC/action/citation_signature","submit_replication":"https://pith.science/pith/VA65YLLAYW2YLQV674X2SOPLRC/action/replication_record"}},"created_at":"2026-07-05T08:52:45.368736+00:00","updated_at":"2026-07-05T08:52:45.368736+00:00"}