{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:V64TH4WAG37T4UTVS2UPS7DN4C","short_pith_number":"pith:V64TH4WA","schema_version":"1.0","canonical_sha256":"afb933f2c036ff3e527596a8f97c6de0a69170e6aeced477f5ba63db4e076367","source":{"kind":"arxiv","id":"2412.12153","version":2},"attestation_state":"computed","paper":{"title":"Revisiting Weight Averaging for Model Merging","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Chanhyuk Lee, Donggyun Kim, Jiho Choi, Seunghoon Hong","submitted_at":"2024-12-11T06:29:20Z","abstract_excerpt":"Model merging aims to build a multi-task learner by combining the parameters of individually fine-tuned models without additional training. While a straightforward approach is to average model parameters across tasks, this often results in suboptimal performance due to interference among parameters across tasks. In this paper, we present intriguing results that weight averaging implicitly induces task vectors centered around the weight averaging itself and that applying a low-rank approximation to these centered task vectors significantly improves merging performance. Our analysis shows that c"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.12153","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2024-12-11T06:29:20Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"ff7de3d0cba0aef35f6ea370fe544e0af2f4cfbcac55cd79bb8beb8de4d12553","abstract_canon_sha256":"f027d66357fae5468cc0a3294fd89d4e0886881f3665b683e4d0da38debe9efb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:43:42.016561Z","signature_b64":"qoNLnGE9Ay8Q1kojWkqpD+L8Ee2/U24XgtZDsa5bQ6Q/pMfK1ww+m+vIy4VdpWmQpTM8tplhFu1rI/6wk1z+BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"afb933f2c036ff3e527596a8f97c6de0a69170e6aeced477f5ba63db4e076367","last_reissued_at":"2026-07-05T10:43:42.015649Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:43:42.015649Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Revisiting Weight Averaging for Model Merging","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Chanhyuk Lee, Donggyun Kim, Jiho Choi, Seunghoon Hong","submitted_at":"2024-12-11T06:29:20Z","abstract_excerpt":"Model merging aims to build a multi-task learner by combining the parameters of individually fine-tuned models without additional training. While a straightforward approach is to average model parameters across tasks, this often results in suboptimal performance due to interference among parameters across tasks. In this paper, we present intriguing results that weight averaging implicitly induces task vectors centered around the weight averaging itself and that applying a low-rank approximation to these centered task vectors significantly improves merging performance. Our analysis shows that c"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.12153","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.12153/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.12153","created_at":"2026-07-05T10:43:42.015707+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.12153v2","created_at":"2026-07-05T10:43:42.015707+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.12153","created_at":"2026-07-05T10:43:42.015707+00:00"},{"alias_kind":"pith_short_12","alias_value":"V64TH4WAG37T","created_at":"2026-07-05T10:43:42.015707+00:00"},{"alias_kind":"pith_short_16","alias_value":"V64TH4WAG37T4UTV","created_at":"2026-07-05T10:43:42.015707+00:00"},{"alias_kind":"pith_short_8","alias_value":"V64TH4WA","created_at":"2026-07-05T10:43:42.015707+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.16533","citing_title":"Kairos: A Regret-Aware Native World-Action Model Stack for Physical AI","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03723","citing_title":"Compress then Merge: From Multiple LoRAs into One Low-Rank Adapter","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2603.02945","citing_title":"ACE-Merging: Data-Free Model Merging with Adaptive Covariance Estimation","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12935","citing_title":"Task Alignment: A Simple Proxy for Practical Model Merging Across Diverse Vision Tasks","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/V64TH4WAG37T4UTVS2UPS7DN4C","json":"https://pith.science/pith/V64TH4WAG37T4UTVS2UPS7DN4C.json","graph_json":"https://pith.science/api/pith-number/V64TH4WAG37T4UTVS2UPS7DN4C/graph.json","events_json":"https://pith.science/api/pith-number/V64TH4WAG37T4UTVS2UPS7DN4C/events.json","paper":"https://pith.science/paper/V64TH4WA"},"agent_actions":{"view_html":"https://pith.science/pith/V64TH4WAG37T4UTVS2UPS7DN4C","download_json":"https://pith.science/pith/V64TH4WAG37T4UTVS2UPS7DN4C.json","view_paper":"https://pith.science/paper/V64TH4WA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.12153&json=true","fetch_graph":"https://pith.science/api/pith-number/V64TH4WAG37T4UTVS2UPS7DN4C/graph.json","fetch_events":"https://pith.science/api/pith-number/V64TH4WAG37T4UTVS2UPS7DN4C/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/V64TH4WAG37T4UTVS2UPS7DN4C/action/timestamp_anchor","attest_storage":"https://pith.science/pith/V64TH4WAG37T4UTVS2UPS7DN4C/action/storage_attestation","attest_author":"https://pith.science/pith/V64TH4WAG37T4UTVS2UPS7DN4C/action/author_attestation","sign_citation":"https://pith.science/pith/V64TH4WAG37T4UTVS2UPS7DN4C/action/citation_signature","submit_replication":"https://pith.science/pith/V64TH4WAG37T4UTVS2UPS7DN4C/action/replication_record"}},"created_at":"2026-07-05T10:43:42.015707+00:00","updated_at":"2026-07-05T10:43:42.015707+00:00"}