{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ZBCMBJ4X3OD4LVANZCHPEGIWZC","short_pith_number":"pith:ZBCMBJ4X","schema_version":"1.0","canonical_sha256":"c844c0a797db87c5d40dc88ef21916c896c7e896e697e5c593f613bf12a68a5c","source":{"kind":"arxiv","id":"2501.01230","version":3},"attestation_state":"computed","paper":{"title":"Modeling Multi-Task Model Merging as Adaptive Projective Gradient Descent","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Anke Tang, Chun Yuan, Li Shen, Xiaochun Cao, Yongxian Wei, Zixuan Hu","submitted_at":"2025-01-02T12:45:21Z","abstract_excerpt":"Merging multiple expert models offers a promising approach for performing multi-task learning without accessing their original data. Existing methods attempt to alleviate task conflicts by sparsifying task vectors or promoting orthogonality among them. However, they overlook the fundamental target of model merging: the merged model performs as closely as possible to task-specific models on respective tasks. We find these methods inevitably discard task-specific information that, while causing conflicts, is crucial for performance. Based on our findings, we frame model merging as a constrained "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.01230","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2025-01-02T12:45:21Z","cross_cats_sorted":[],"title_canon_sha256":"b6cedc8f3281a6a99038a0e2b25ad9bcdb4f71ef37d16dd497692ecdd4fa018c","abstract_canon_sha256":"f0fabe74dc327ce09335c1cf4dc81e46dfa926a831fb450ba1a60f9c3237cb50"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:09:44.483906Z","signature_b64":"1zXMbrONUuGtno82FldQPmwTn+0aXZV+PY6Edg7wrUadwmE86kmLBeBlg1uf/euM5PlnWEptY7/cyNA2lO1VCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c844c0a797db87c5d40dc88ef21916c896c7e896e697e5c593f613bf12a68a5c","last_reissued_at":"2026-07-05T11:09:44.483433Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:09:44.483433Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Modeling Multi-Task Model Merging as Adaptive Projective Gradient Descent","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Anke Tang, Chun Yuan, Li Shen, Xiaochun Cao, Yongxian Wei, Zixuan Hu","submitted_at":"2025-01-02T12:45:21Z","abstract_excerpt":"Merging multiple expert models offers a promising approach for performing multi-task learning without accessing their original data. Existing methods attempt to alleviate task conflicts by sparsifying task vectors or promoting orthogonality among them. However, they overlook the fundamental target of model merging: the merged model performs as closely as possible to task-specific models on respective tasks. We find these methods inevitably discard task-specific information that, while causing conflicts, is crucial for performance. Based on our findings, we frame model merging as a constrained "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.01230","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.01230/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.01230","created_at":"2026-07-05T11:09:44.483488+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.01230v3","created_at":"2026-07-05T11:09:44.483488+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.01230","created_at":"2026-07-05T11:09:44.483488+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZBCMBJ4X3OD4","created_at":"2026-07-05T11:09:44.483488+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZBCMBJ4X3OD4LVAN","created_at":"2026-07-05T11:09:44.483488+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZBCMBJ4X","created_at":"2026-07-05T11:09:44.483488+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18627","citing_title":"PACT: Preserving Anchored Cores in Task-vectors for Model Merging","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01689","citing_title":"Model Merging as Probabilistic Inference in Fine-Tuning Parameter Space","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03391","citing_title":"When Model Merging Breaks Routing: Training-Free Calibration for MoE","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20948","citing_title":"Memory Grafting: Scaling Language Model Pre-training via Offline Conditional Memory","ref_index":50,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZBCMBJ4X3OD4LVANZCHPEGIWZC","json":"https://pith.science/pith/ZBCMBJ4X3OD4LVANZCHPEGIWZC.json","graph_json":"https://pith.science/api/pith-number/ZBCMBJ4X3OD4LVANZCHPEGIWZC/graph.json","events_json":"https://pith.science/api/pith-number/ZBCMBJ4X3OD4LVANZCHPEGIWZC/events.json","paper":"https://pith.science/paper/ZBCMBJ4X"},"agent_actions":{"view_html":"https://pith.science/pith/ZBCMBJ4X3OD4LVANZCHPEGIWZC","download_json":"https://pith.science/pith/ZBCMBJ4X3OD4LVANZCHPEGIWZC.json","view_paper":"https://pith.science/paper/ZBCMBJ4X","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.01230&json=true","fetch_graph":"https://pith.science/api/pith-number/ZBCMBJ4X3OD4LVANZCHPEGIWZC/graph.json","fetch_events":"https://pith.science/api/pith-number/ZBCMBJ4X3OD4LVANZCHPEGIWZC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZBCMBJ4X3OD4LVANZCHPEGIWZC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZBCMBJ4X3OD4LVANZCHPEGIWZC/action/storage_attestation","attest_author":"https://pith.science/pith/ZBCMBJ4X3OD4LVANZCHPEGIWZC/action/author_attestation","sign_citation":"https://pith.science/pith/ZBCMBJ4X3OD4LVANZCHPEGIWZC/action/citation_signature","submit_replication":"https://pith.science/pith/ZBCMBJ4X3OD4LVANZCHPEGIWZC/action/replication_record"}},"created_at":"2026-07-05T11:09:44.483488+00:00","updated_at":"2026-07-05T11:09:44.483488+00:00"}