{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:J6JLDDTHDFIQGGHSL37YCGZYAL","short_pith_number":"pith:J6JLDDTH","schema_version":"1.0","canonical_sha256":"4f92b18e6719510318f25eff811b3802ee12d47f7a9cf505b3efeda003cb8476","source":{"kind":"arxiv","id":"2410.03617","version":1},"attestation_state":"computed","paper":{"title":"What Matters for Model Merging at Scale?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Alexandra Chronopoulou, Jonathan Lai, Manaal Faruqui, Mohit Bansal, Prateek Yadav, Tsendsuren Munkhdalai, Tu Vu","submitted_at":"2024-10-04T17:17:19Z","abstract_excerpt":"Model merging aims to combine multiple expert models into a more capable single model, offering benefits such as reduced storage and serving costs, improved generalization, and support for decentralized model development. Despite its promise, previous studies have primarily focused on merging a few small models. This leaves many unanswered questions about the effect of scaling model size and how it interplays with other key factors -- like the base model quality and number of expert models -- , to affect the merged model's performance. This work systematically evaluates the utility of model me"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.03617","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-04T17:17:19Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"969f6eedaf6662e6cd46125dbfee51e3653145dddba404c487afa2c24ca486e2","abstract_canon_sha256":"5497a1f80f20bc31b665079a9528907e428fb4e9c37375dcdfb347fa11b6593f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:16:03.798321Z","signature_b64":"hb2ZQUhxkihGEve67n4T6H1WN4KSe0bt01hdsW+Y95arOUZwQ+/APx/PyGcplxlvk5zdMmmMuuXDCgBtlobKDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4f92b18e6719510318f25eff811b3802ee12d47f7a9cf505b3efeda003cb8476","last_reissued_at":"2026-07-05T09:16:03.797793Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:16:03.797793Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"What Matters for Model Merging at Scale?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Alexandra Chronopoulou, Jonathan Lai, Manaal Faruqui, Mohit Bansal, Prateek Yadav, Tsendsuren Munkhdalai, Tu Vu","submitted_at":"2024-10-04T17:17:19Z","abstract_excerpt":"Model merging aims to combine multiple expert models into a more capable single model, offering benefits such as reduced storage and serving costs, improved generalization, and support for decentralized model development. Despite its promise, previous studies have primarily focused on merging a few small models. This leaves many unanswered questions about the effect of scaling model size and how it interplays with other key factors -- like the base model quality and number of expert models -- , to affect the merged model's performance. This work systematically evaluates the utility of model me"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.03617","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.03617/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.03617","created_at":"2026-07-05T09:16:03.797869+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.03617v1","created_at":"2026-07-05T09:16:03.797869+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.03617","created_at":"2026-07-05T09:16:03.797869+00:00"},{"alias_kind":"pith_short_12","alias_value":"J6JLDDTHDFIQ","created_at":"2026-07-05T09:16:03.797869+00:00"},{"alias_kind":"pith_short_16","alias_value":"J6JLDDTHDFIQGGHS","created_at":"2026-07-05T09:16:03.797869+00:00"},{"alias_kind":"pith_short_8","alias_value":"J6JLDDTH","created_at":"2026-07-05T09:16:03.797869+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07289","citing_title":"Closed-Form Spectral Regularization for Multi-Task Model Merging","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2509.24244","citing_title":"Model Merging Scaling Laws in Large Language Models","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2510.04686","citing_title":"How does the optimizer implicitly bias the model merging loss landscape?","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2511.01831","citing_title":"Routing-Based Continual Learning for Multimodal Large Language Models","ref_index":70,"is_internal_anchor":false},{"citing_arxiv_id":"2408.07666","citing_title":"Model Merging in LLMs, MLLMs, and Beyond: Methods, Theories, Applications and Opportunities","ref_index":264,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10903","citing_title":"CapVector: Learning Transferable Capability Vectors in Parametric Space for Vision-Language-Action Models","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/J6JLDDTHDFIQGGHSL37YCGZYAL","json":"https://pith.science/pith/J6JLDDTHDFIQGGHSL37YCGZYAL.json","graph_json":"https://pith.science/api/pith-number/J6JLDDTHDFIQGGHSL37YCGZYAL/graph.json","events_json":"https://pith.science/api/pith-number/J6JLDDTHDFIQGGHSL37YCGZYAL/events.json","paper":"https://pith.science/paper/J6JLDDTH"},"agent_actions":{"view_html":"https://pith.science/pith/J6JLDDTHDFIQGGHSL37YCGZYAL","download_json":"https://pith.science/pith/J6JLDDTHDFIQGGHSL37YCGZYAL.json","view_paper":"https://pith.science/paper/J6JLDDTH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.03617&json=true","fetch_graph":"https://pith.science/api/pith-number/J6JLDDTHDFIQGGHSL37YCGZYAL/graph.json","fetch_events":"https://pith.science/api/pith-number/J6JLDDTHDFIQGGHSL37YCGZYAL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/J6JLDDTHDFIQGGHSL37YCGZYAL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/J6JLDDTHDFIQGGHSL37YCGZYAL/action/storage_attestation","attest_author":"https://pith.science/pith/J6JLDDTHDFIQGGHSL37YCGZYAL/action/author_attestation","sign_citation":"https://pith.science/pith/J6JLDDTHDFIQGGHSL37YCGZYAL/action/citation_signature","submit_replication":"https://pith.science/pith/J6JLDDTHDFIQGGHSL37YCGZYAL/action/replication_record"}},"created_at":"2026-07-05T09:16:03.797869+00:00","updated_at":"2026-07-05T09:16:03.797869+00:00"}