{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:DNCMRHB5C2SOMY26LIE3DFBUGZ","short_pith_number":"pith:DNCMRHB5","schema_version":"1.0","canonical_sha256":"1b44c89c3d16a4e6635e5a09b19434367a248bf10d9362cacd4b95120cdf4e4b","source":{"kind":"arxiv","id":"2111.09832","version":2},"attestation_state":"computed","paper":{"title":"Merging Models with Fisher-Weighted Averaging","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Colin Raffel, Michael Matena","submitted_at":"2021-11-18T17:59:35Z","abstract_excerpt":"Averaging the parameters of models that have the same architecture and initialization can provide a means of combining their respective capabilities. In this paper, we take the perspective that this \"merging\" operation can be seen as choosing parameters that approximately maximize the joint likelihood of the posteriors of the models' parameters. Computing a simple average of the models' parameters therefore corresponds to making an isotropic Gaussian approximation to their posteriors. We develop an alternative merging procedure based on the Laplace approximation where we approximate each model"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2111.09832","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-11-18T17:59:35Z","cross_cats_sorted":[],"title_canon_sha256":"6845ce5452125c3f97adbe4211c4f9d6f113311be7b03a5ae9006dcc0f07f340","abstract_canon_sha256":"eab2bd99cdfba1a52f48fb09738fc04278ed5c957135ccd28a0fb19faa6b029c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:51:39.001388Z","signature_b64":"+i+pAPupR+4ZBEu3RLsCun+fLmpHJYpjnh/9JE86QsDlIWuzbMc4d+u+rioLlbJmYDfsj8HunLXidskIoZvbDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1b44c89c3d16a4e6635e5a09b19434367a248bf10d9362cacd4b95120cdf4e4b","last_reissued_at":"2026-07-05T04:51:39.000947Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:51:39.000947Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Merging Models with Fisher-Weighted Averaging","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Colin Raffel, Michael Matena","submitted_at":"2021-11-18T17:59:35Z","abstract_excerpt":"Averaging the parameters of models that have the same architecture and initialization can provide a means of combining their respective capabilities. In this paper, we take the perspective that this \"merging\" operation can be seen as choosing parameters that approximately maximize the joint likelihood of the posteriors of the models' parameters. Computing a simple average of the models' parameters therefore corresponds to making an isotropic Gaussian approximation to their posteriors. We develop an alternative merging procedure based on the Laplace approximation where we approximate each model"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2111.09832","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2111.09832/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2111.09832","created_at":"2026-07-05T04:51:39.001004+00:00"},{"alias_kind":"arxiv_version","alias_value":"2111.09832v2","created_at":"2026-07-05T04:51:39.001004+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2111.09832","created_at":"2026-07-05T04:51:39.001004+00:00"},{"alias_kind":"pith_short_12","alias_value":"DNCMRHB5C2SO","created_at":"2026-07-05T04:51:39.001004+00:00"},{"alias_kind":"pith_short_16","alias_value":"DNCMRHB5C2SOMY26","created_at":"2026-07-05T04:51:39.001004+00:00"},{"alias_kind":"pith_short_8","alias_value":"DNCMRHB5","created_at":"2026-07-05T04:51:39.001004+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23591","citing_title":"Quantifying the Agreement Between Data-Influence and Data-Similarity to Understand LLM Behavior","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2408.01119","citing_title":"Task Prompt Vectors: Effective Initialization through Multi-Task Soft-Prompt Transfer","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20296","citing_title":"Spectral Unforgetting: Post-Hoc Recovery of Damaged Capabilities Without Retraining","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2511.01831","citing_title":"Routing-Based Continual Learning for Multimodal Large Language Models","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2212.04089","citing_title":"Editing Models with Task Arithmetic","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07244","citing_title":"Experience Sharing in Mutual Reinforcement Learning for Heterogeneous Language Models","ref_index":58,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DNCMRHB5C2SOMY26LIE3DFBUGZ","json":"https://pith.science/pith/DNCMRHB5C2SOMY26LIE3DFBUGZ.json","graph_json":"https://pith.science/api/pith-number/DNCMRHB5C2SOMY26LIE3DFBUGZ/graph.json","events_json":"https://pith.science/api/pith-number/DNCMRHB5C2SOMY26LIE3DFBUGZ/events.json","paper":"https://pith.science/paper/DNCMRHB5"},"agent_actions":{"view_html":"https://pith.science/pith/DNCMRHB5C2SOMY26LIE3DFBUGZ","download_json":"https://pith.science/pith/DNCMRHB5C2SOMY26LIE3DFBUGZ.json","view_paper":"https://pith.science/paper/DNCMRHB5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2111.09832&json=true","fetch_graph":"https://pith.science/api/pith-number/DNCMRHB5C2SOMY26LIE3DFBUGZ/graph.json","fetch_events":"https://pith.science/api/pith-number/DNCMRHB5C2SOMY26LIE3DFBUGZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DNCMRHB5C2SOMY26LIE3DFBUGZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DNCMRHB5C2SOMY26LIE3DFBUGZ/action/storage_attestation","attest_author":"https://pith.science/pith/DNCMRHB5C2SOMY26LIE3DFBUGZ/action/author_attestation","sign_citation":"https://pith.science/pith/DNCMRHB5C2SOMY26LIE3DFBUGZ/action/citation_signature","submit_replication":"https://pith.science/pith/DNCMRHB5C2SOMY26LIE3DFBUGZ/action/replication_record"}},"created_at":"2026-07-05T04:51:39.001004+00:00","updated_at":"2026-07-05T04:51:39.001004+00:00"}