{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:B46B3LAUYSXAMGEQRAMCWQJNOK","short_pith_number":"pith:B46B3LAU","schema_version":"1.0","canonical_sha256":"0f3c1dac14c4ae06189088182b412d728b155881770888ec46c614d315a87fef","source":{"kind":"arxiv","id":"2408.07990","version":1},"attestation_state":"computed","paper":{"title":"FuseChat: Knowledge Fusion of Chat Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Fanqi Wan, Longguang Zhong, Ruijun Chen, Xiaojun Quan, Ziyi Yang","submitted_at":"2024-08-15T07:37:24Z","abstract_excerpt":"While training large language models (LLMs) from scratch can indeed lead to models with distinct capabilities and strengths, it incurs substantial costs and may lead to redundancy in competencies. Knowledge fusion aims to integrate existing LLMs of diverse architectures and capabilities into a more potent LLM through lightweight continual training, thereby reducing the need for costly LLM development. In this work, we propose a new framework for the knowledge fusion of chat LLMs through two main stages, resulting in FuseChat. Firstly, we conduct pairwise knowledge fusion on source chat LLMs of"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.07990","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-08-15T07:37:24Z","cross_cats_sorted":[],"title_canon_sha256":"18e04e7f1855da768fd2a3ce9ca68eb75d2b6f8eaf8eca08d497e9884a97c381","abstract_canon_sha256":"0dd5e125adf81458ff63add3d983baa8991be8748ecda19d83cb07ea65ce0925"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:55:42.926176Z","signature_b64":"izvNZrpn2DZ627ixVQ/GekyOccVY95m/Vs8rKO83BOZkqFp3+wQaCwQi69HPMnqZmopmGnfdU0NE1TdWlDlKDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0f3c1dac14c4ae06189088182b412d728b155881770888ec46c614d315a87fef","last_reissued_at":"2026-07-05T08:55:42.925773Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:55:42.925773Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FuseChat: Knowledge Fusion of Chat Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Fanqi Wan, Longguang Zhong, Ruijun Chen, Xiaojun Quan, Ziyi Yang","submitted_at":"2024-08-15T07:37:24Z","abstract_excerpt":"While training large language models (LLMs) from scratch can indeed lead to models with distinct capabilities and strengths, it incurs substantial costs and may lead to redundancy in competencies. Knowledge fusion aims to integrate existing LLMs of diverse architectures and capabilities into a more potent LLM through lightweight continual training, thereby reducing the need for costly LLM development. In this work, we propose a new framework for the knowledge fusion of chat LLMs through two main stages, resulting in FuseChat. Firstly, we conduct pairwise knowledge fusion on source chat LLMs of"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.07990","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.07990/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.07990","created_at":"2026-07-05T08:55:42.925829+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.07990v1","created_at":"2026-07-05T08:55:42.925829+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.07990","created_at":"2026-07-05T08:55:42.925829+00:00"},{"alias_kind":"pith_short_12","alias_value":"B46B3LAUYSXA","created_at":"2026-07-05T08:55:42.925829+00:00"},{"alias_kind":"pith_short_16","alias_value":"B46B3LAUYSXAMGEQ","created_at":"2026-07-05T08:55:42.925829+00:00"},{"alias_kind":"pith_short_8","alias_value":"B46B3LAU","created_at":"2026-07-05T08:55:42.925829+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03391","citing_title":"When Model Merging Breaks Routing: Training-Free Calibration for MoE","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2505.13893","citing_title":"InfiGFusion: Graph-on-Logits Distillation via Efficient Gromov-Wasserstein for Model Fusion","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2604.01674","citing_title":"Can Heterogeneous Language Models Be Fused?","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2510.04265","citing_title":"Don't Pass@k: A Bayesian Framework for Large Language Model Evaluation","ref_index":98,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06377","citing_title":"The Master Key Hypothesis: Unlocking Cross-Model Capability Transfer via Linear Subspace Alignment","ref_index":72,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/B46B3LAUYSXAMGEQRAMCWQJNOK","json":"https://pith.science/pith/B46B3LAUYSXAMGEQRAMCWQJNOK.json","graph_json":"https://pith.science/api/pith-number/B46B3LAUYSXAMGEQRAMCWQJNOK/graph.json","events_json":"https://pith.science/api/pith-number/B46B3LAUYSXAMGEQRAMCWQJNOK/events.json","paper":"https://pith.science/paper/B46B3LAU"},"agent_actions":{"view_html":"https://pith.science/pith/B46B3LAUYSXAMGEQRAMCWQJNOK","download_json":"https://pith.science/pith/B46B3LAUYSXAMGEQRAMCWQJNOK.json","view_paper":"https://pith.science/paper/B46B3LAU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.07990&json=true","fetch_graph":"https://pith.science/api/pith-number/B46B3LAUYSXAMGEQRAMCWQJNOK/graph.json","fetch_events":"https://pith.science/api/pith-number/B46B3LAUYSXAMGEQRAMCWQJNOK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/B46B3LAUYSXAMGEQRAMCWQJNOK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/B46B3LAUYSXAMGEQRAMCWQJNOK/action/storage_attestation","attest_author":"https://pith.science/pith/B46B3LAUYSXAMGEQRAMCWQJNOK/action/author_attestation","sign_citation":"https://pith.science/pith/B46B3LAUYSXAMGEQRAMCWQJNOK/action/citation_signature","submit_replication":"https://pith.science/pith/B46B3LAUYSXAMGEQRAMCWQJNOK/action/replication_record"}},"created_at":"2026-07-05T08:55:42.925829+00:00","updated_at":"2026-07-05T08:55:42.925829+00:00"}