{"as_of":"2026-08-10T08:59:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:3e1b4ab8d37bc9e20deeaf9863ae59107a6bb5895fd4dc49addd1296eac1cb12","coverage":[{"denominator":38,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":38,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T13:44:36.927598Z","state":"measured"},{"denominator":43,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":43,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":5,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":5,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-01T02:42:45.039852Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-06-30T11:24:38.144959Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"cited_work":{"arxiv_id":"2505.21226","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.21226","snapshot_observed_at":"2026-06-30T11:24:38.144959Z","title":"Why do more experts fail? a theoretical analysis of model merging","venue":null,"work_id":"746e96a3-5cdb-4287-9fca-4a6ebbbc4006","year":2025},"citing_paper":{"arxiv_id":"2408.07666","last_updated":"2025-12-31T04:06:49Z","snapshot_observed_at":"2026-08-07T23:28:24.025478Z","submitted_at":"2024-08-14T16:58:48Z","title":"Model Merging in LLMs, MLLMs, and Beyond: Methods, Theories, Applications and Opportunities","version":5},"reference_index":245,"source":"pdf_text","source_observed_at":"2026-05-17T22:16:04.386706Z"},"links":{"cited_paper":"/paper/2505.21226","citing_paper":"/paper/2408.07666"},"observation_digest":"sha256:bd3e9d36bc86c1a8b6d42cfdbc716925a7f01549e58a6b21a4883d87263cc258","observation_id":"6d716525-a76d-431b-822c-a9ccb32d56a8","resolution":{"observed_at":"2026-05-17T22:16:04.730252Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"cited_work":{"arxiv_id":"2505.21226","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.21226","snapshot_observed_at":"2026-06-30T11:24:38.144959Z","title":"Why do more experts fail? a theoretical analysis of model merging","venue":null,"work_id":"746e96a3-5cdb-4287-9fca-4a6ebbbc4006","year":2025},"citing_paper":{"arxiv_id":"2605.12960","last_updated":"2026-05-20T10:12:11Z","snapshot_observed_at":"2026-07-06T23:24:37.915639Z","submitted_at":"2026-05-13T03:50:54Z","title":"DiM\\textsuperscript{3}: Bridging Multilingual and Multimodal Models via Direction- and Magnitude-Aware Merging","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-14T20:24:30.679219Z"},"links":{"cited_paper":"/paper/2505.21226","citing_paper":"/paper/2605.12960"},"observation_digest":"sha256:a310099c7889c95fe4663ad59d84831fe1ad09f225c1c71557ae8a0f70dc851c","observation_id":"26ade7fe-5909-4180-85f5-c4b39f0e4243","resolution":{"observed_at":"2026-05-14T20:42:58.884086Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"cited_work":{"arxiv_id":"2505.21226","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.21226","snapshot_observed_at":"2026-06-30T11:24:38.144959Z","title":"Why do more experts fail? a theoretical analysis of model merging","venue":null,"work_id":"746e96a3-5cdb-4287-9fca-4a6ebbbc4006","year":2025},"citing_paper":{"arxiv_id":"2605.12960","last_updated":"2026-05-20T10:12:11Z","snapshot_observed_at":"2026-07-06T23:24:37.915639Z","submitted_at":"2026-05-13T03:50:54Z","title":"DiM\\textsuperscript{3}: Bridging Multilingual and Multimodal Models via Direction- and Magnitude-Aware Merging","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-21T09:12:22.712240Z"},"links":{"cited_paper":"/paper/2505.21226","citing_paper":"/paper/2605.12960"},"observation_digest":"sha256:899f87f1ad15c046af8622ecd130fad034728c20fb920c6a07334b730969c7e1","observation_id":"ad5d758d-41b7-48d4-a036-7e2ace2d781f","resolution":{"observed_at":"2026-05-21T09:14:05.733226Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"cited_work":{"arxiv_id":"2505.21226","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.21226","snapshot_observed_at":"2026-06-30T11:24:38.144959Z","title":"Why do more experts fail? a theoretical analysis of model merging","venue":null,"work_id":"746e96a3-5cdb-4287-9fca-4a6ebbbc4006","year":2025},"citing_paper":{"arxiv_id":"2606.28373","last_updated":"2026-06-17T12:39:10Z","snapshot_observed_at":"2026-07-07T00:02:29.000757Z","submitted_at":"2026-06-17T12:39:10Z","title":"Model Merging to Evolution: Parameter Space Exploration for Expert Models","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-06-30T11:20:04.579617Z"},"links":{"cited_paper":"/paper/2505.21226","citing_paper":"/paper/2606.28373"},"observation_digest":"sha256:1c45ce109899150ca042653c84eedbe4c51381d97cdd7852713af38b0e47524d","observation_id":"2d3685b4-69b6-40b0-bf5d-a0e5f3951412","resolution":{"observed_at":"2026-06-30T11:24:38.146287Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.21226","snapshot_observed_at":"2026-08-01T02:42:45.039852Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.25366","last_updated":"2026-07-28T07:17:44Z","snapshot_observed_at":"2026-08-09T03:33:30.122090Z","submitted_at":"2026-07-28T07:17:44Z","title":"Sharpness-aware Model Merging with Salience Recovery for LLM-based Cross-Domain Sequential Recommendation","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-01T02:42:45.039852Z"},"links":{"cited_paper":"/paper/2505.21226","citing_paper":"/paper/2607.25366"},"observation_digest":"sha256:354f5e33052d6d95b935e7b2398dee79548785ae3dd14912a1e4262394e2a811","observation_id":"dc62f227-3112-43cc-a5f8-3d2b3c7ff4d5","resolution":{"observed_at":"2026-08-01T02:42:45.039852Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2505.21226/citation-record","integrity":"/paper/2505.21226/integrity","json":"/paper/2505.21226/citation-record.json","paper":"/paper/2505.21226"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:44:32.608750Z","title":"Evolutionary optimization of model merging recipes.Nature Machine Intelligence, pages 1–10, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:32.608750Z"},"links":{"citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:18387bcb2e89ee20b9c5913046e1afc6d3cfeee72bd96bfef208143a247e5a1c","observation_id":"e29d4997-1560-4b85-9b1a-a27a5b968a5e","resolution":{"observed_at":"2026-08-07T13:44:32.608750Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:44:38.561248Z","title":"Living on the edge: Phase transitions in convex programs with random data.Information and Inference: A Journal of the IMA, 3(3):224–294, 2014","venue":null,"work_id":"527d5f0f-cdac-45b8-8b4a-3f4a113f5aed","year":2014},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:32.677745Z"},"links":{"citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:7f125b531c3025b0ed0c6e4ecd47f201284007e03a568d2afaaafcac3e6552f5","observation_id":"9a59c3fa-eed0-4c63-975b-6ee3caa393bc","resolution":{"observed_at":"2026-08-07T13:44:38.665865Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2108.07732","last_updated":"2021-08-16T03:57:30Z","snapshot_observed_at":"2026-08-02T19:23:53.535075Z","submitted_at":"2021-08-16T03:57:30Z","title":"Program Synthesis with Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2108.07732","snapshot_observed_at":"2026-08-07T13:44:32.790393Z","title":"Program synthesis with large language models.arXiv preprint arXiv:2108.07732, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:32.790393Z"},"links":{"cited_paper":"/paper/2108.07732","citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:970f8dce3fc09bd39dd6b36a904e56c24083ddf97fda757196e42cd7ccdb61d4","observation_id":"5b7ccb7a-ad41-4bfe-bc71-3eea40db3453","resolution":{"observed_at":"2026-08-07T13:44:32.790393Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.12153","last_updated":"2025-04-03T11:46:20Z","snapshot_observed_at":"2026-07-06T20:08:02.963967Z","submitted_at":"2024-12-11T06:29:20Z","title":"Revisiting Weight Averaging for Model Merging","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.12153","snapshot_observed_at":"2026-08-07T13:44:32.908553Z","title":"Revisiting weight averaging for model merging.arXiv preprint arXiv:2412.12153, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:32.908553Z"},"links":{"cited_paper":"/paper/2412.12153","citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:036b58eef17015924778466d276a70599a67486a85ab8dc0ff80d2306b76093e","observation_id":"d265a83f-900c-46d6-ba8d-a30de0a616bc","resolution":{"observed_at":"2026-08-07T13:44:32.908553Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.11163","last_updated":"2025-05-31T23:27:47Z","snapshot_observed_at":"2026-07-31T07:44:00.085049Z","submitted_at":"2024-10-15T00:59:17Z","title":"Model Swarms: Collaborative Search to Adapt LLM Experts via Swarm Intelligence","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.11163","snapshot_observed_at":"2026-08-07T13:44:33.011741Z","title":"Model swarms: Col- laborative search to adapt llm experts via swarm intelligence.arXiv preprint arXiv:2410.11163, 2024","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:33.011741Z"},"links":{"cited_paper":"/paper/2410.11163","citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:dc931571d85498dcfc09b24218c31d4006e88b5175316746e40186b3c04e2a47","observation_id":"733253d7-b3d9-40ee-b755-673c64c1964c","resolution":{"observed_at":"2026-08-07T13:44:33.011741Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2009.03300","last_updated":"2021-01-12T18:57:11Z","snapshot_observed_at":"2026-08-09T10:28:06.906299Z","submitted_at":"2020-09-07T17:59:25Z","title":"Measuring Massive Multitask Language Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2009.03300","snapshot_observed_at":"2026-08-07T13:44:33.131628Z","title":"Measuring massive multitask language understanding.arXiv preprint arXiv:2009.03300, 2020","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:33.131628Z"},"links":{"cited_paper":"/paper/2009.03300","citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:044a80ae1e0091b309105ed3024957fd0da127d77f7673e429294bb66aae47c6","observation_id":"ffdab5c8-5420-4a83-8fa6-a02dcf3303ea","resolution":{"observed_at":"2026-08-07T13:44:33.131628Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2103.03874","last_updated":"2021-11-08T21:30:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-03-05T18:59:39Z","title":"Measuring Mathematical Problem Solving With the MATH Dataset","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2103.03874","snapshot_observed_at":"2026-08-07T13:44:33.275322Z","title":"Measuring mathematical problem solving with the math dataset","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:33.275322Z"},"links":{"cited_paper":"/paper/2103.03874","citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:a0e3287ba3deb3fd9ff401ad82d68d5becc063488e9a3703be6265f1d2fcdd43","observation_id":"d5213ade-951a-4a9c-aa43-dddaed611e6b","resolution":{"observed_at":"2026-08-07T13:44:33.275322Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:44:33.398276Z","title":"Lora: Low-rank adaptation of large language models.ICLR, 1 (2):3, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:33.398276Z"},"links":{"citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:d126064a2570d7f9c37e10d875d87c65a54b17e530242ee7d44ef18951512f68","observation_id":"7de861e7-b0a6-4fa2-b594-db4a575157f0","resolution":{"observed_at":"2026-08-07T13:44:33.398276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.13269","last_updated":"2024-08-19T03:31:19Z","snapshot_observed_at":"2026-07-06T15:58:08.777051Z","submitted_at":"2023-07-25T05:39:21Z","title":"LoraHub: Efficient Cross-Task Generalization via Dynamic LoRA Composition","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.13269","snapshot_observed_at":"2026-08-07T13:44:33.469848Z","title":"Lo- rahub: Efficient cross-task generalization via dynamic lora composition.arXiv preprint arXiv:2307.13269, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:33.469848Z"},"links":{"cited_paper":"/paper/2307.13269","citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:8772919b2f38a4726c8c2b2fa1a326d6c26fe5d514e0a63cbfe227ba9af2f909","observation_id":"f0422f54-d67a-40e6-af3e-8cd41f517f1e","resolution":{"observed_at":"2026-08-07T13:44:33.469848Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.10702","last_updated":"2023-11-20T02:01:33Z","snapshot_observed_at":"2026-08-09T03:30:59.643714Z","submitted_at":"2023-11-17T18:45:45Z","title":"Camels in a Changing Climate: Enhancing LM Adaptation with Tulu 2","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.10702","snapshot_observed_at":"2026-08-07T13:44:33.633023Z","title":"Camels in a changing climate: Enhancing lm adaptation with tulu 2.arXiv preprint arXiv:2311.10702, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:33.633023Z"},"links":{"cited_paper":"/paper/2311.10702","citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:7d1b4575ada2cd2245165c3b4863c794e975ffeb35ac259727d971d0368239fc","observation_id":"9ca7eda8-fd42-4db0-af8b-0cf7fd2f754f","resolution":{"observed_at":"2026-08-07T13:44:33.633023Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:44:33.713425Z","title":"The singular value decomposition: Its computation and some applications.IEEE Transactions on automatic control, 25(2):164–176, 1980","venue":null,"work_id":null,"year":1980},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:33.713425Z"},"links":{"citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:ecb0ef12be9fab4ddb6562f7b7fedfbd603d2efed759cb632445c5681361a01a","observation_id":"2088241f-d960-4be2-a8f4-5cdd46cdb00d","resolution":{"observed_at":"2026-08-07T13:44:33.713425Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2107.05802","last_updated":"2022-02-03T06:16:05Z","snapshot_observed_at":"2026-08-09T03:32:23.627884Z","submitted_at":"2021-07-13T01:29:24Z","title":"How many degrees of freedom do we need to train deep networks: a loss landscape perspective","version":2},"cited_work":{"arxiv_id":"2107.05802","doi":null,"metadata_source":"pith","pith_arxiv_id":"2107.05802","snapshot_observed_at":"2026-08-07T13:44:37.263502Z","title":"How many degrees of freedom do we need to train deep networks: a loss landscape perspective","venue":"cs.LG","work_id":"47010271-0eef-4cb7-8a6c-e1c1ea384f37","year":2021},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:33.847753Z"},"links":{"cited_paper":"/paper/2107.05802","citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:615970f36f86e147b8781d6d9ee3f76c5ef6a99422bb15d1bc55fd8d5b3f3742","observation_id":"45931957-3294-4814-9ff7-98b9280b7feb","resolution":{"observed_at":"2026-08-07T13:44:37.342378Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.08691","last_updated":"2021-09-02T17:34:41Z","snapshot_observed_at":"2026-08-06T15:24:34.790850Z","submitted_at":"2021-04-18T03:19:26Z","title":"The Power of Scale for Parameter-Efficient Prompt Tuning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.08691","snapshot_observed_at":"2026-08-07T13:44:33.957436Z","title":"The power of scale for parameter-efficient prompt tuning.arXiv preprint arXiv:2104.08691, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:33.957436Z"},"links":{"cited_paper":"/paper/2104.08691","citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:473dfe65d598823f00f8af63dce2ab97a30b75a947cbde633c142fc919790567","observation_id":"44f2716b-30b3-49e4-b820-8e7757dcc472","resolution":{"observed_at":"2026-08-07T13:44:33.957436Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2101.00190","last_updated":"2021-01-01T08:00:36Z","snapshot_observed_at":"2026-07-06T10:29:18.734092Z","submitted_at":"2021-01-01T08:00:36Z","title":"Prefix-Tuning: Optimizing Continuous Prompts for Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2101.00190","snapshot_observed_at":"2026-08-07T13:44:34.061461Z","title":"Prefix-tuning: Optimizing continuous prompts for generation","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:34.061461Z"},"links":{"cited_paper":"/paper/2101.00190","citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:a52d1958ce80eb35e0d676a4396fb7da6254bffb22ceb20bbf04303474ea3a57","observation_id":"7eb2afd5-079c-4563-99bd-828a43b0c9fd","resolution":{"observed_at":"2026-08-07T13:44:34.061461Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:44:34.182206Z","title":"Gpt understands, too.AI Open, 5:208–215, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:34.182206Z"},"links":{"citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:3d0f35bc3eb0193972d0665b8c4fdcd907717705aa0420864b5425938008eafb","observation_id":"7c15703c-d038-4a43-a856-03f26bf24f44","resolution":{"observed_at":"2026-08-07T13:44:34.182206Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.15207","last_updated":"2024-06-17T10:35:06Z","snapshot_observed_at":"2026-07-06T17:21:10.236126Z","submitted_at":"2024-01-26T21:14:32Z","title":"HiFT: A Hierarchical Full Parameter Fine-Tuning Strategy","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.15207","snapshot_observed_at":"2026-08-07T13:44:34.329800Z","title":"Hift: A hierarchical full parameter fine-tuning strategy.arXiv preprint arXiv:2401.15207, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:34.329800Z"},"links":{"cited_paper":"/paper/2401.15207","citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:73a85cfc369cc5bbfe4c88e00f77d08749712a47bb73c6ce76475d23a7cfaf67","observation_id":"1c215081-916c-4983-9e2e-78cf062b1f82","resolution":{"observed_at":"2026-08-07T13:44:34.329800Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.12851","last_updated":"2024-02-20T09:30:48Z","snapshot_observed_at":"2026-07-06T17:32:44.565172Z","submitted_at":"2024-02-20T09:30:48Z","title":"MoELoRA: Contrastive Learning Guided Mixture of Experts on Parameter-Efficient Fine-Tuning for Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.12851","snapshot_observed_at":"2026-08-07T13:44:34.475689Z","title":"Moelora: Contrastive learning guided mixture of experts on parameter-efficient fine-tuning for large language models.arXiv preprint arXiv:2402.12851, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:34.475689Z"},"links":{"cited_paper":"/paper/2402.12851","citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:ddda64b80ae1bc858fa6423407fc583ae5f5e6cb02b16e0cf3648a46ff09f9f5","observation_id":"7a8a99c8-b520-4f7e-bd11-4cfa72266b36","resolution":{"observed_at":"2026-08-07T13:44:34.475689Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.11531","last_updated":"2024-04-17T16:24:07Z","snapshot_observed_at":"2026-07-06T18:01:40.563108Z","submitted_at":"2024-04-17T16:24:07Z","title":"Pack of LLMs: Model Fusion at Test-Time via Perplexity Optimization","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.11531","snapshot_observed_at":"2026-08-07T13:44:34.591129Z","title":"Pack of llms: Model fusion at test-time via perplexity optimization.arXiv preprint arXiv:2404.11531, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:34.591129Z"},"links":{"cited_paper":"/paper/2404.11531","citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:5a67b7d8b2129969bbb0bf57f888c2c1087dd7f2c8654e3037a905395b25ee64","observation_id":"965601e2-00c2-4dbe-b1b5-8a9b2d8b731d","resolution":{"observed_at":"2026-08-07T13:44:34.591129Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:44:34.686559Z","title":"Orthogonal adaptation for modular customization of diffusion models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:34.686559Z"},"links":{"citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:a298cd0e97ad44bbbcd171f97f335041bff6e45670153bdcd3ea6ac3d02d76a7","observation_id":"fdbba087-fced-4277-9e75-5854c8769165","resolution":{"observed_at":"2026-08-07T13:44:34.686559Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:44:38.384990Z","title":"Ensemble learning.Ensemble machine learning: Methods and applications, pages 1–34, 2012","venue":null,"work_id":"ac0b8e23-d9a0-4949-8b07-1b87503bf001","year":2012},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:34.813147Z"},"links":{"citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:5a142a3098e4eeeda4a2fc472a79f9a194d861e3073219941e735bc842e7e28b","observation_id":"3ed1964a-6fa9-44ec-8605-31cba3b67616","resolution":{"observed_at":"2026-08-07T13:44:38.461282Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:44:34.956683Z","title":"Acceleration of stochastic approximation by averaging","venue":null,"work_id":null,"year":1992},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:34.956683Z"},"links":{"citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:9b56745139b3848bc79eca25a5f4912c6c184ff0b521680c28970ca180c66a84","observation_id":"ae21ea70-f297-4238-ad6c-fc313818264b","resolution":{"observed_at":"2026-08-07T13:44:34.956683Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.13025","last_updated":"2024-12-02T06:40:50Z","snapshot_observed_at":"2026-08-10T02:33:42.583525Z","submitted_at":"2024-10-16T20:33:06Z","title":"LoRA Soups: Merging LoRAs for Practical Skill Composition Tasks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.13025","snapshot_observed_at":"2026-08-07T13:44:35.077153Z","title":"Lora soups: Merging loras for practical skill composition tasks.arXiv preprint arXiv:2410.13025, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:35.077153Z"},"links":{"cited_paper":"/paper/2410.13025","citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:b2ee254b541498eefad2a2bafcb5f7a1eab6c8f8969f4b237ddf8a321c821d24","observation_id":"28bdaed8-72f2-4824-90a6-750e342590e2","resolution":{"observed_at":"2026-08-07T13:44:35.077153Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:44:35.176600Z","title":"Rewarded soups: towards pareto-optimal alignment by interpolating weights fine-tuned on diverse rewards.Advances in Neural Information Processing Systems, 36:71095–71134, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:35.176600Z"},"links":{"citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:645165ae6d1fe7674b28dc88157b59c1181c1f9c2c73111f5f39f6b80f928dca","observation_id":"0130fcf2-28b3-47cb-a2ed-04fdd78c52a3","resolution":{"observed_at":"2026-08-07T13:44:35.176600Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.03057","last_updated":"2022-10-06T17:03:34Z","snapshot_observed_at":"2026-07-06T14:00:37.875054Z","submitted_at":"2022-10-06T17:03:34Z","title":"Language Models are Multilingual Chain-of-Thought Reasoners","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.03057","snapshot_observed_at":"2026-08-07T13:44:35.245526Z","title":"Language models are multilingual chain-of-thought reasoners.arXiv preprint arXiv:2210.03057, 2022","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:35.245526Z"},"links":{"cited_paper":"/paper/2210.03057","citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:ca46f40cdd1053602a197b46220a0343b5fd0e00930fb0bd18a07fec8859dfe3","observation_id":"54698c76-c14f-44bb-9fa6-b99da878fa47","resolution":{"observed_at":"2026-08-07T13:44:35.245526Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1811.00937","last_updated":"2019-03-15T18:02:58Z","snapshot_observed_at":"2026-08-10T07:37:55.457401Z","submitted_at":"2018-11-02T15:34:29Z","title":"CommonsenseQA: A Question Answering Challenge Targeting Commonsense Knowledge","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1811.00937","snapshot_observed_at":"2026-08-07T13:44:35.393197Z","title":"Commonsenseqa: A ques- tion answering challenge targeting commonsense knowledge.arXiv preprint arXiv:1811.00937, 2018","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:35.393197Z"},"links":{"cited_paper":"/paper/1811.00937","citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:0333bf30471557320aa72d50442da51e3fe23965b5f6b904ae87cbd8cfd9240c","observation_id":"b91491c9-ebee-4694-bcdc-98a6d2ef8eea","resolution":{"observed_at":"2026-08-07T13:44:35.393197Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:44:38.203090Z","title":"Unlocking the potential of model merging for low-resource languages","venue":null,"work_id":"d7d5add1-21b9-4ac0-b36c-0c7ebac7a0f3","year":2024},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:35.501545Z"},"links":{"citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:706b1c86925e0f5c805d8b8fbf35d6db4cdf52f8277052ac39e0bc3c76d15e82","observation_id":"edba27e7-7af2-41c2-8b25-c472e8ae6012","resolution":{"observed_at":"2026-08-07T13:44:38.282317Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00118","last_updated":"2024-10-02T15:22:49Z","snapshot_observed_at":"2026-08-02T16:20:09.773989Z","submitted_at":"2024-07-31T19:13:07Z","title":"Gemma 2: Improving Open Language Models at a Practical Size","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00118","snapshot_observed_at":"2026-08-07T13:44:35.591163Z","title":"Gemma 2: Improving open language models at a practical size.arXiv preprint arXiv:2408.00118, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:35.591163Z"},"links":{"cited_paper":"/paper/2408.00118","citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:7d7e098d434d4b28247bc13b90882ff4add89714e06334685703e25617252a14","observation_id":"2ee1fe58-fe4d-47ae-994f-d2886fcc51d7","resolution":{"observed_at":"2026-08-07T13:44:35.591163Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:44:38.056049Z","title":"Estimation in high dimensions: a geometric perspective","venue":null,"work_id":"66abdc70-c1cb-41ed-b5c8-a4f6728480a3","year":2015},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:35.688556Z"},"links":{"citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:b7e496726f94fa5a9df24680edca98c236bdc830975c856a5a1e84ba17f13b38","observation_id":"a78f64a5-2263-436b-97be-29d03aaa3b5a","resolution":{"observed_at":"2026-08-07T13:44:38.115696Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:44:37.869154Z","title":"Principal component analysis.Chemometrics and intelligent laboratory systems, 2(1-3):37–52, 1987","venue":null,"work_id":"06090968-bc47-40ec-a81e-c410de24de5b","year":1987},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:35.866143Z"},"links":{"citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:dd3ec0415e700eb0d64f047a059a00c7d92dcde1ac0b12ae4ec58c9422b4f9aa","observation_id":"6038e34c-c053-442d-8403-119bcdca394e","resolution":{"observed_at":"2026-08-07T13:44:37.954937Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:44:35.949478Z","title":"Ties-merging: Resolving interference when merging models.Advances in Neural Information Processing Systems, 36:7093–7115, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:35.949478Z"},"links":{"citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:e2ca0b0c772e902bfe3f71c63d1808aae264bdce31a570a35489dd93818e097f","observation_id":"713dba1e-8030-4202-a7d1-c82ead5ede4c","resolution":{"observed_at":"2026-08-07T13:44:35.949478Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03617","last_updated":"2024-10-04T17:17:19Z","snapshot_observed_at":"2026-07-06T19:27:52.298292Z","submitted_at":"2024-10-04T17:17:19Z","title":"What Matters for Model Merging at Scale?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.03617","snapshot_observed_at":"2026-08-07T13:44:36.048280Z","title":"What matters for model merging at scale?arXiv preprint arXiv:2410.03617, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:36.048280Z"},"links":{"cited_paper":"/paper/2410.03617","citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:b3c87a7a6be159012fff5385ee4ce8c97dc6dedcd496d8f3833e98d68e9a1269","observation_id":"d75f9e00-ade1-4bfa-87a9-8e15c2c7303c","resolution":{"observed_at":"2026-08-07T13:44:36.048280Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02575","last_updated":"2024-05-28T06:42:31Z","snapshot_observed_at":"2026-08-10T08:38:34.106077Z","submitted_at":"2023-10-04T04:26:33Z","title":"AdaMerging: Adaptive Model Merging for Multi-Task Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.02575","snapshot_observed_at":"2026-08-07T13:44:36.149510Z","title":"Adamerging: Adaptive model merging for multi-task learning.arXiv preprint arXiv:2310.02575, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:36.149510Z"},"links":{"cited_paper":"/paper/2310.02575","citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:1f45f70acb7487b7c3740afee8bc9eac51fe515203eefeaf8cc054c2b9151bda","observation_id":"b7b36f27-715f-4f4d-8a43-5462faa7c56e","resolution":{"observed_at":"2026-08-07T13:44:36.149510Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:44:36.280683Z","title":"Language models are super mario: Absorbing abilities from homologous models as a free lunch","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:36.280683Z"},"links":{"citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:750fe72cde6f0bfdd2ae3e80d2bf4b5a8b81735927f8914e6c3b11389eab942e","observation_id":"e76b9d63-4e76-4c68-8ac7-2615fb08f8dd","resolution":{"observed_at":"2026-08-07T13:44:36.280683Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:44:37.682900Z","title":"Emotion detection on tv show transcripts with sequence- based convolutional neural networks","venue":null,"work_id":"42fc08de-d456-45f6-860f-2686a3d202f8","year":2018},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:36.447501Z"},"links":{"citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:d67e647a32697b78dd5300dd160b5a4fe6355d6581557da1ffdbb6d1e1e72c24","observation_id":"40992453-269c-43b9-915f-861568f81fa5","resolution":{"observed_at":"2026-08-07T13:44:37.766416Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:44:36.543531Z","title":"Composing parameter-efficient modules with arithmetic operation.Advances in Neural Information Processing Systems, 36:12589–12610, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:36.543531Z"},"links":{"citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:e13f84107425d9471bb6e23388c87f827df019b476981e841ba46fb085b9287d","observation_id":"3ecc682e-1cf1-4fd6-bd56-cee4ab56e020","resolution":{"observed_at":"2026-08-07T13:44:36.543531Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01155","last_updated":"2025-03-03T04:03:31Z","snapshot_observed_at":"2026-08-07T17:34:57.455440Z","submitted_at":"2025-03-03T04:03:31Z","title":"Nature-Inspired Population-Based Evolution of Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.01155","snapshot_observed_at":"2026-08-07T13:44:36.640223Z","title":"Nature-inspired population-based evolution of large language models.arXiv preprint arXiv:2503.01155, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:36.640223Z"},"links":{"cited_paper":"/paper/2503.01155","citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:43c4def2819b8567fe5b6e1bc2e94fe96ce11fc106e77222469f08a5c2fb125d","observation_id":"e151f6e4-f5a9-48ce-8208-125b4414a467","resolution":{"observed_at":"2026-08-07T13:44:36.640223Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.13372","last_updated":"2024-06-27T22:44:48Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-03-20T08:08:54Z","title":"LlamaFactory: Unified Efficient Fine-Tuning of 100+ Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.13372","snapshot_observed_at":"2026-08-07T13:44:36.776858Z","title":"Llamafactory: Unified efficient fine-tuning of 100+ language models.arXiv preprint arXiv:2403.13372, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:36.776858Z"},"links":{"cited_paper":"/paper/2403.13372","citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:07ec3ef3cc348db4db4b68a5e2c4ca2ef41d4d09ecd1cf29861cae49e900b52f","observation_id":"ae8bc4e2-e34d-47ee-aacc-744cc08cdf91","resolution":{"observed_at":"2026-08-07T13:44:36.776858Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:44:37.513153Z","title":"particle","venue":null,"work_id":"2a7e689a-d839-476a-8e4b-7f19553e2aa6","year":2021},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:36.927598Z"},"links":{"citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:665ee78d30ac532746c0824c0574fa6d4a67292e9fbf86c7a6c890094411aa35","observation_id":"b5c8c531-4c32-486b-8dce-fc8bdfb86d45","resolution":{"observed_at":"2026-08-07T13:44:37.554033Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","latest_version":2,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging"},"reference_resolution":{"displayed":38,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":30,"verified_exact":1,"verified_fuzzy":7},"total_outbound_references":38},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 38 of 38 outbound references and 5 inbound Pith citation observations for arXiv:2505.21226."}