{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:DPCVE2B4ZQBKTSRIYGYN5XND3N","short_pith_number":"pith:DPCVE2B4","schema_version":"1.0","canonical_sha256":"1bc552683ccc02a9ca28c1b0dedda3db7d11b8146912c68a01f3ce5dbe481dad","source":{"kind":"arxiv","id":"2410.12937","version":1},"attestation_state":"computed","paper":{"title":"Merge to Learn: Efficiently Adding Skills to Language Models with Model Merging","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Hannaneh Hajishirzi, Jacob Morrison, Jesse Dodge, Noah A. Smith, Pang Wei Koh, Pradeep Dasigi","submitted_at":"2024-10-16T18:23:50Z","abstract_excerpt":"Adapting general-purpose language models to new skills is currently an expensive process that must be repeated as new instruction datasets targeting new skills are created, or can cause the models to forget older skills. In this work, we investigate the effectiveness of adding new skills to preexisting models by training on the new skills in isolation and later merging with the general model (e.g. using task vectors). In experiments focusing on scientific literature understanding, safety, and coding, we find that the parallel-train-then-merge procedure, which is significantly cheaper than retr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.12937","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-16T18:23:50Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"91bca69d2e6877f2f8c01b8bbfca9ff4c5fb194ab4149e878a0fbdd415fb4af5","abstract_canon_sha256":"a1c97e9c4c2fe3fad97a57418d45e223695fb1e589384298598ac42970772801"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:21:53.005823Z","signature_b64":"J6O4EKEZ+J00wL78MuCQADeqCscRpWIhE5sQRjGrd5lVFYsWIjpFfIK2oAzsE/Q7lckj/R9DLDsDMhUqaeQrBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1bc552683ccc02a9ca28c1b0dedda3db7d11b8146912c68a01f3ce5dbe481dad","last_reissued_at":"2026-07-05T09:21:53.005366Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:21:53.005366Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Merge to Learn: Efficiently Adding Skills to Language Models with Model Merging","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Hannaneh Hajishirzi, Jacob Morrison, Jesse Dodge, Noah A. Smith, Pang Wei Koh, Pradeep Dasigi","submitted_at":"2024-10-16T18:23:50Z","abstract_excerpt":"Adapting general-purpose language models to new skills is currently an expensive process that must be repeated as new instruction datasets targeting new skills are created, or can cause the models to forget older skills. In this work, we investigate the effectiveness of adding new skills to preexisting models by training on the new skills in isolation and later merging with the general model (e.g. using task vectors). In experiments focusing on scientific literature understanding, safety, and coding, we find that the parallel-train-then-merge procedure, which is significantly cheaper than retr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.12937","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.12937/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.12937","created_at":"2026-07-05T09:21:53.005422+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.12937v1","created_at":"2026-07-05T09:21:53.005422+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.12937","created_at":"2026-07-05T09:21:53.005422+00:00"},{"alias_kind":"pith_short_12","alias_value":"DPCVE2B4ZQBK","created_at":"2026-07-05T09:21:53.005422+00:00"},{"alias_kind":"pith_short_16","alias_value":"DPCVE2B4ZQBKTSRI","created_at":"2026-07-05T09:21:53.005422+00:00"},{"alias_kind":"pith_short_8","alias_value":"DPCVE2B4","created_at":"2026-07-05T09:21:53.005422+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.18473","citing_title":"Train Separately, Merge Together: Modular Post-Training with Mixture-of-Experts","ref_index":29,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DPCVE2B4ZQBKTSRIYGYN5XND3N","json":"https://pith.science/pith/DPCVE2B4ZQBKTSRIYGYN5XND3N.json","graph_json":"https://pith.science/api/pith-number/DPCVE2B4ZQBKTSRIYGYN5XND3N/graph.json","events_json":"https://pith.science/api/pith-number/DPCVE2B4ZQBKTSRIYGYN5XND3N/events.json","paper":"https://pith.science/paper/DPCVE2B4"},"agent_actions":{"view_html":"https://pith.science/pith/DPCVE2B4ZQBKTSRIYGYN5XND3N","download_json":"https://pith.science/pith/DPCVE2B4ZQBKTSRIYGYN5XND3N.json","view_paper":"https://pith.science/paper/DPCVE2B4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.12937&json=true","fetch_graph":"https://pith.science/api/pith-number/DPCVE2B4ZQBKTSRIYGYN5XND3N/graph.json","fetch_events":"https://pith.science/api/pith-number/DPCVE2B4ZQBKTSRIYGYN5XND3N/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DPCVE2B4ZQBKTSRIYGYN5XND3N/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DPCVE2B4ZQBKTSRIYGYN5XND3N/action/storage_attestation","attest_author":"https://pith.science/pith/DPCVE2B4ZQBKTSRIYGYN5XND3N/action/author_attestation","sign_citation":"https://pith.science/pith/DPCVE2B4ZQBKTSRIYGYN5XND3N/action/citation_signature","submit_replication":"https://pith.science/pith/DPCVE2B4ZQBKTSRIYGYN5XND3N/action/replication_record"}},"created_at":"2026-07-05T09:21:53.005422+00:00","updated_at":"2026-07-05T09:21:53.005422+00:00"}