{"as_of":"2026-08-13T23:13:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:6d22917be9d795c9e0501482c3eb835ab9f4f4d87e924e1eeaf5dbf3150911c8","coverage":[{"denominator":38,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":38,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T13:44:36.927598Z","state":"measured"},{"denominator":43,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":43,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-13T06:32:02.005865+00:00","state":"measured"},{"denominator":5,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":5,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-01T02:42:45.039852Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-06-30T11:24:38.144959Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"cited_work":{"arxiv_id":"2505.21226","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.21226","snapshot_observed_at":"2026-06-30T11:24:38.144959Z","title":"Why do more experts fail? a theoretical analysis of model merging","venue":null,"work_id":"746e96a3-5cdb-4287-9fca-4a6ebbbc4006","year":2025},"citing_paper":{"arxiv_id":"2408.07666","last_updated":"2025-12-31T04:06:49Z","snapshot_observed_at":"2026-08-07T23:28:24.025478Z","submitted_at":"2024-08-14T16:58:48Z","title":"Model Merging in LLMs, MLLMs, and Beyond: Methods, Theories, Applications and Opportunities","version":5},"reference_index":245,"source":"pdf_text","source_observed_at":"2026-05-17T22:16:04.386706Z"},"links":{"cited_paper":"/paper/2505.21226","citing_paper":"/paper/2408.07666"},"observation_digest":"sha256:9d59d7dee6a302c102ff6fa9fb101336a626eaa879e8cd04e29dde9f6be30a4a","observation_id":"6d716525-a76d-431b-822c-a9ccb32d56a8","resolution":{"observed_at":"2026-05-17T22:16:04.730252Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"cited_work":{"arxiv_id":"2505.21226","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.21226","snapshot_observed_at":"2026-06-30T11:24:38.144959Z","title":"Why do more experts fail? a theoretical analysis of model merging","venue":null,"work_id":"746e96a3-5cdb-4287-9fca-4a6ebbbc4006","year":2025},"citing_paper":{"arxiv_id":"2605.12960","last_updated":"2026-05-20T10:12:11Z","snapshot_observed_at":"2026-08-11T11:09:08.684005Z","submitted_at":"2026-05-13T03:50:54Z","title":"DiM\\textsuperscript{3}: Bridging Multilingual and Multimodal Models via Direction- and Magnitude-Aware Merging","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-14T20:24:30.679219Z"},"links":{"cited_paper":"/paper/2505.21226","citing_paper":"/paper/2605.12960"},"observation_digest":"sha256:2255a6d78671bfe45467cd1e11fbc5480608d25f3c52142e70156b7abee69ae8","observation_id":"26ade7fe-5909-4180-85f5-c4b39f0e4243","resolution":{"observed_at":"2026-05-14T20:42:58.884086Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"cited_work":{"arxiv_id":"2505.21226","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.21226","snapshot_observed_at":"2026-06-30T11:24:38.144959Z","title":"Why do more experts fail? a theoretical analysis of model merging","venue":null,"work_id":"746e96a3-5cdb-4287-9fca-4a6ebbbc4006","year":2025},"citing_paper":{"arxiv_id":"2605.12960","last_updated":"2026-05-20T10:12:11Z","snapshot_observed_at":"2026-08-11T11:09:08.684005Z","submitted_at":"2026-05-13T03:50:54Z","title":"DiM\\textsuperscript{3}: Bridging Multilingual and Multimodal Models via Direction- and Magnitude-Aware Merging","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-21T09:12:22.712240Z"},"links":{"cited_paper":"/paper/2505.21226","citing_paper":"/paper/2605.12960"},"observation_digest":"sha256:db9d10536f0432a9ca97e510fe3cf02b40bc197111b4a4808fcfd6bf88ea62ab","observation_id":"ad5d758d-41b7-48d4-a036-7e2ace2d781f","resolution":{"observed_at":"2026-05-21T09:14:05.733226Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"cited_work":{"arxiv_id":"2505.21226","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.21226","snapshot_observed_at":"2026-06-30T11:24:38.144959Z","title":"Why do more experts fail? a theoretical analysis of model merging","venue":null,"work_id":"746e96a3-5cdb-4287-9fca-4a6ebbbc4006","year":2025},"citing_paper":{"arxiv_id":"2606.28373","last_updated":"2026-06-17T12:39:10Z","snapshot_observed_at":"2026-08-13T05:13:36.518448Z","submitted_at":"2026-06-17T12:39:10Z","title":"Model Merging to Evolution: Parameter Space Exploration for Expert Models","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-06-30T11:20:04.579617Z"},"links":{"cited_paper":"/paper/2505.21226","citing_paper":"/paper/2606.28373"},"observation_digest":"sha256:7ef1768050b7c058fdc03739cb13de5bd7e908ce34bf342806f195639c6647a9","observation_id":"2d3685b4-69b6-40b0-bf5d-a0e5f3951412","resolution":{"observed_at":"2026-06-30T11:24:38.146287Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.21226","snapshot_observed_at":"2026-08-01T02:42:45.039852Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.25366","last_updated":"2026-07-28T07:17:44Z","snapshot_observed_at":"2026-08-09T03:33:30.122090Z","submitted_at":"2026-07-28T07:17:44Z","title":"Sharpness-aware Model Merging with Salience Recovery for LLM-based Cross-Domain Sequential Recommendation","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-01T02:42:45.039852Z"},"links":{"cited_paper":"/paper/2505.21226","citing_paper":"/paper/2607.25366"},"observation_digest":"sha256:354f5e33052d6d95b935e7b2398dee79548785ae3dd14912a1e4262394e2a811","observation_id":"dc62f227-3112-43cc-a5f8-3d2b3c7ff4d5","resolution":{"observed_at":"2026-08-01T02:42:45.039852Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2505.21226/citation-record","integrity":"/paper/2505.21226/integrity","json":"/paper/2505.21226/citation-record.json","paper":"/paper/2505.21226"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:44:32.608750Z","title":"Evolutionary optimization of model merging recipes.Nature Machine Intelligence, pages 1–10, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:32.608750Z"},"links":{"citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:18387bcb2e89ee20b9c5913046e1afc6d3cfeee72bd96bfef208143a247e5a1c","observation_id":"e29d4997-1560-4b85-9b1a-a27a5b968a5e","resolution":{"observed_at":"2026-08-07T13:44:32.608750Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:44:38.561248Z","title":"Living on the edge: Phase transitions in convex programs with random data.Information and Inference: A Journal of the IMA, 3(3):224–294, 2014","venue":null,"work_id":"527d5f0f-cdac-45b8-8b4a-3f4a113f5aed","year":2014},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:32.677745Z"},"links":{"citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:8a60834b9eb9b2cd7b9a85dd634c79d576b12076f461ec398a5b9eecfa156bb9","observation_id":"9a59c3fa-eed0-4c63-975b-6ee3caa393bc","resolution":{"observed_at":"2026-08-07T13:44:38.665865Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2108.07732","last_updated":"2021-08-16T03:57:30Z","snapshot_observed_at":"2026-08-02T19:23:53.535075Z","submitted_at":"2021-08-16T03:57:30Z","title":"Program Synthesis with Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2108.07732","snapshot_observed_at":"2026-08-07T13:44:32.790393Z","title":"Program synthesis with large language models.arXiv preprint arXiv:2108.07732, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:32.790393Z"},"links":{"cited_paper":"/paper/2108.07732","citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:970f8dce3fc09bd39dd6b36a904e56c24083ddf97fda757196e42cd7ccdb61d4","observation_id":"5b7ccb7a-ad41-4bfe-bc71-3eea40db3453","resolution":{"observed_at":"2026-08-07T13:44:32.790393Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.12153","last_updated":"2025-04-03T11:46:20Z","snapshot_observed_at":"2026-08-11T18:08:47.719255Z","submitted_at":"2024-12-11T06:29:20Z","title":"Revisiting Weight Averaging for Model Merging","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.12153","snapshot_observed_at":"2026-08-07T13:44:32.908553Z","title":"Revisiting weight averaging for model merging.arXiv preprint arXiv:2412.12153, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:32.908553Z"},"links":{"cited_paper":"/paper/2412.12153","citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:ba97d2653077083381db36dd358e820a9e73efff6a85cc463d73d3d8d6ac98d2","observation_id":"d265a83f-900c-46d6-ba8d-a30de0a616bc","resolution":{"observed_at":"2026-08-07T13:44:32.908553Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.11163","last_updated":"2025-05-31T23:27:47Z","snapshot_observed_at":"2026-08-12T22:23:11.893662Z","submitted_at":"2024-10-15T00:59:17Z","title":"Model Swarms: Collaborative Search to Adapt LLM Experts via Swarm Intelligence","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.11163","snapshot_observed_at":"2026-08-07T13:44:33.011741Z","title":"Model swarms: Col- laborative search to adapt llm experts via swarm intelligence.arXiv preprint arXiv:2410.11163, 2024","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:33.011741Z"},"links":{"cited_paper":"/paper/2410.11163","citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:7126bc67bc772750b3edb3d8f72d95fe508799764f6aa3d63283db1e14ddb6d7","observation_id":"733253d7-b3d9-40ee-b755-673c64c1964c","resolution":{"observed_at":"2026-08-07T13:44:33.011741Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2009.03300","last_updated":"2021-01-12T18:57:11Z","snapshot_observed_at":"2026-08-13T20:44:28.824685Z","submitted_at":"2020-09-07T17:59:25Z","title":"Measuring Massive Multitask Language Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2009.03300","snapshot_observed_at":"2026-08-07T13:44:33.131628Z","title":"Measuring massive multitask language understanding.arXiv preprint arXiv:2009.03300, 2020","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:33.131628Z"},"links":{"cited_paper":"/paper/2009.03300","citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:22d85c0a71a41a96854c18eb92dbb89090b1c9eacf12fad8094c602886c030d7","observation_id":"ffdab5c8-5420-4a83-8fa6-a02dcf3303ea","resolution":{"observed_at":"2026-08-07T13:44:33.131628Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2103.03874","last_updated":"2021-11-08T21:30:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-03-05T18:59:39Z","title":"Measuring Mathematical Problem Solving With the MATH Dataset","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2103.03874","snapshot_observed_at":"2026-08-07T13:44:33.275322Z","title":"Measuring mathematical problem solving with the math dataset","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:33.275322Z"},"links":{"cited_paper":"/paper/2103.03874","citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:a0e3287ba3deb3fd9ff401ad82d68d5becc063488e9a3703be6265f1d2fcdd43","observation_id":"d5213ade-951a-4a9c-aa43-dddaed611e6b","resolution":{"observed_at":"2026-08-07T13:44:33.275322Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:44:33.398276Z","title":"Lora: Low-rank adaptation of large language models.ICLR, 1 (2):3, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:33.398276Z"},"links":{"citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:d126064a2570d7f9c37e10d875d87c65a54b17e530242ee7d44ef18951512f68","observation_id":"7de861e7-b0a6-4fa2-b594-db4a575157f0","resolution":{"observed_at":"2026-08-07T13:44:33.398276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.13269","last_updated":"2024-08-19T03:31:19Z","snapshot_observed_at":"2026-08-13T19:38:58.092030Z","submitted_at":"2023-07-25T05:39:21Z","title":"LoraHub: Efficient Cross-Task Generalization via Dynamic LoRA Composition","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.13269","snapshot_observed_at":"2026-08-07T13:44:33.469848Z","title":"Lo- rahub: Efficient cross-task generalization via dynamic lora composition.arXiv preprint arXiv:2307.13269, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:33.469848Z"},"links":{"cited_paper":"/paper/2307.13269","citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:7b140ae3cbd4afaa7286ed26d9169039eb549c83673e3e50ca74fd13aba8522c","observation_id":"f0422f54-d67a-40e6-af3e-8cd41f517f1e","resolution":{"observed_at":"2026-08-07T13:44:33.469848Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.10702","last_updated":"2023-11-20T02:01:33Z","snapshot_observed_at":"2026-08-13T05:22:58.620371Z","submitted_at":"2023-11-17T18:45:45Z","title":"Camels in a Changing Climate: Enhancing LM Adaptation with Tulu 2","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.10702","snapshot_observed_at":"2026-08-07T13:44:33.633023Z","title":"Camels in a changing climate: Enhancing lm adaptation with tulu 2.arXiv preprint arXiv:2311.10702, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:33.633023Z"},"links":{"cited_paper":"/paper/2311.10702","citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:bc0fb9e0207ffe7df7a673a2c898cc5cc918717e93b04a2f5fb46a50c13e5a88","observation_id":"9ca7eda8-fd42-4db0-af8b-0cf7fd2f754f","resolution":{"observed_at":"2026-08-07T13:44:33.633023Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:44:33.713425Z","title":"The singular value decomposition: Its computation and some applications.IEEE Transactions on automatic control, 25(2):164–176, 1980","venue":null,"work_id":null,"year":1980},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:33.713425Z"},"links":{"citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:ecb0ef12be9fab4ddb6562f7b7fedfbd603d2efed759cb632445c5681361a01a","observation_id":"2088241f-d960-4be2-a8f4-5cdd46cdb00d","resolution":{"observed_at":"2026-08-07T13:44:33.713425Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2107.05802","last_updated":"2022-02-03T06:16:05Z","snapshot_observed_at":"2026-08-09T03:32:23.627884Z","submitted_at":"2021-07-13T01:29:24Z","title":"How many degrees of freedom do we need to train deep networks: a loss landscape perspective","version":2},"cited_work":{"arxiv_id":"2107.05802","doi":null,"metadata_source":"pith","pith_arxiv_id":"2107.05802","snapshot_observed_at":"2026-08-07T13:44:37.263502Z","title":"How many degrees of freedom do we need to train deep networks: a loss landscape perspective","venue":"cs.LG","work_id":"47010271-0eef-4cb7-8a6c-e1c1ea384f37","year":2021},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:33.847753Z"},"links":{"cited_paper":"/paper/2107.05802","citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:9b7ab5144294cf1a5714e6187fc20952cfc44335215de3aefeecee4b25f40f74","observation_id":"45931957-3294-4814-9ff7-98b9280b7feb","resolution":{"observed_at":"2026-08-07T13:44:37.342378Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.08691","last_updated":"2021-09-02T17:34:41Z","snapshot_observed_at":"2026-08-06T15:24:34.790850Z","submitted_at":"2021-04-18T03:19:26Z","title":"The Power of Scale for Parameter-Efficient Prompt Tuning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.08691","snapshot_observed_at":"2026-08-07T13:44:33.957436Z","title":"The power of scale for parameter-efficient prompt tuning.arXiv preprint arXiv:2104.08691, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:33.957436Z"},"links":{"cited_paper":"/paper/2104.08691","citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:473dfe65d598823f00f8af63dce2ab97a30b75a947cbde633c142fc919790567","observation_id":"44f2716b-30b3-49e4-b820-8e7757dcc472","resolution":{"observed_at":"2026-08-07T13:44:33.957436Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2101.00190","last_updated":"2021-01-01T08:00:36Z","snapshot_observed_at":"2026-08-12T12:37:28.207761Z","submitted_at":"2021-01-01T08:00:36Z","title":"Prefix-Tuning: Optimizing Continuous Prompts for Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2101.00190","snapshot_observed_at":"2026-08-07T13:44:34.061461Z","title":"Prefix-tuning: Optimizing continuous prompts for generation","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:34.061461Z"},"links":{"cited_paper":"/paper/2101.00190","citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:ee73ee21a2a855dd523c3e3bd4d3f5e6819901615b43f7e3305271cfb0a6442a","observation_id":"7eb2afd5-079c-4563-99bd-828a43b0c9fd","resolution":{"observed_at":"2026-08-07T13:44:34.061461Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:44:34.182206Z","title":"Gpt understands, too.AI Open, 5:208–215, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:34.182206Z"},"links":{"citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:3d0f35bc3eb0193972d0665b8c4fdcd907717705aa0420864b5425938008eafb","observation_id":"7c15703c-d038-4a43-a856-03f26bf24f44","resolution":{"observed_at":"2026-08-07T13:44:34.182206Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.15207","last_updated":"2024-06-17T10:35:06Z","snapshot_observed_at":"2026-08-13T04:33:46.557585Z","submitted_at":"2024-01-26T21:14:32Z","title":"HiFT: A Hierarchical Full Parameter Fine-Tuning Strategy","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.15207","snapshot_observed_at":"2026-08-07T13:44:34.329800Z","title":"Hift: A hierarchical full parameter fine-tuning strategy.arXiv preprint arXiv:2401.15207, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:34.329800Z"},"links":{"cited_paper":"/paper/2401.15207","citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:5420050d7d618fedac7186bf368f1eb18a81bbd9c5a7134592b4b6c3012310f1","observation_id":"1c215081-916c-4983-9e2e-78cf062b1f82","resolution":{"observed_at":"2026-08-07T13:44:34.329800Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.12851","last_updated":"2024-02-20T09:30:48Z","snapshot_observed_at":"2026-08-13T04:14:41.180426Z","submitted_at":"2024-02-20T09:30:48Z","title":"MoELoRA: Contrastive Learning Guided Mixture of Experts on Parameter-Efficient Fine-Tuning for Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.12851","snapshot_observed_at":"2026-08-07T13:44:34.475689Z","title":"Moelora: Contrastive learning guided mixture of experts on parameter-efficient fine-tuning for large language models.arXiv preprint arXiv:2402.12851, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:34.475689Z"},"links":{"cited_paper":"/paper/2402.12851","citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:aa6a3a4ef9ee7f6fbf0d84f865ae3ac6d217d61e24ee6939103d2839d133a0ce","observation_id":"7a8a99c8-b520-4f7e-bd11-4cfa72266b36","resolution":{"observed_at":"2026-08-07T13:44:34.475689Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.11531","last_updated":"2024-04-17T16:24:07Z","snapshot_observed_at":"2026-08-13T00:28:03.597400Z","submitted_at":"2024-04-17T16:24:07Z","title":"Pack of LLMs: Model Fusion at Test-Time via Perplexity Optimization","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.11531","snapshot_observed_at":"2026-08-07T13:44:34.591129Z","title":"Pack of llms: Model fusion at test-time via perplexity optimization.arXiv preprint arXiv:2404.11531, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:34.591129Z"},"links":{"cited_paper":"/paper/2404.11531","citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:b606e38903f5a724942fc4032ead2ab8eebc66d7cbb343d34d202a85e1d5dba0","observation_id":"965601e2-00c2-4dbe-b1b5-8a9b2d8b731d","resolution":{"observed_at":"2026-08-07T13:44:34.591129Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:44:34.686559Z","title":"Orthogonal adaptation for modular customization of diffusion models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:34.686559Z"},"links":{"citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:a298cd0e97ad44bbbcd171f97f335041bff6e45670153bdcd3ea6ac3d02d76a7","observation_id":"fdbba087-fced-4277-9e75-5854c8769165","resolution":{"observed_at":"2026-08-07T13:44:34.686559Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:44:38.384990Z","title":"Ensemble learning.Ensemble machine learning: Methods and applications, pages 1–34, 2012","venue":null,"work_id":"ac0b8e23-d9a0-4949-8b07-1b87503bf001","year":2012},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:34.813147Z"},"links":{"citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:45ddeb08e35288e7461126f14fca6e4dbeed24586b8329a5c8077a5a067188b3","observation_id":"3ed1964a-6fa9-44ec-8605-31cba3b67616","resolution":{"observed_at":"2026-08-07T13:44:38.461282Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:44:34.956683Z","title":"Acceleration of stochastic approximation by averaging","venue":null,"work_id":null,"year":1992},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:34.956683Z"},"links":{"citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:9b56745139b3848bc79eca25a5f4912c6c184ff0b521680c28970ca180c66a84","observation_id":"ae21ea70-f297-4238-ad6c-fc313818264b","resolution":{"observed_at":"2026-08-07T13:44:34.956683Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.13025","last_updated":"2024-12-02T06:40:50Z","snapshot_observed_at":"2026-08-12T22:21:29.853385Z","submitted_at":"2024-10-16T20:33:06Z","title":"LoRA Soups: Merging LoRAs for Practical Skill Composition Tasks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.13025","snapshot_observed_at":"2026-08-07T13:44:35.077153Z","title":"Lora soups: Merging loras for practical skill composition tasks.arXiv preprint arXiv:2410.13025, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:35.077153Z"},"links":{"cited_paper":"/paper/2410.13025","citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:1ed9026df0162a55578e0317d72661726a4c66422f0ba5895a1c4d4aec6fc54a","observation_id":"28bdaed8-72f2-4824-90a6-750e342590e2","resolution":{"observed_at":"2026-08-07T13:44:35.077153Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:44:35.176600Z","title":"Rewarded soups: towards pareto-optimal alignment by interpolating weights fine-tuned on diverse rewards.Advances in Neural Information Processing Systems, 36:71095–71134, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:35.176600Z"},"links":{"citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:645165ae6d1fe7674b28dc88157b59c1181c1f9c2c73111f5f39f6b80f928dca","observation_id":"0130fcf2-28b3-47cb-a2ed-04fdd78c52a3","resolution":{"observed_at":"2026-08-07T13:44:35.176600Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.03057","last_updated":"2022-10-06T17:03:34Z","snapshot_observed_at":"2026-07-06T14:00:37.875054Z","submitted_at":"2022-10-06T17:03:34Z","title":"Language Models are Multilingual Chain-of-Thought Reasoners","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.03057","snapshot_observed_at":"2026-08-07T13:44:35.245526Z","title":"Language models are multilingual chain-of-thought reasoners.arXiv preprint arXiv:2210.03057, 2022","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:35.245526Z"},"links":{"cited_paper":"/paper/2210.03057","citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:ca46f40cdd1053602a197b46220a0343b5fd0e00930fb0bd18a07fec8859dfe3","observation_id":"54698c76-c14f-44bb-9fa6-b99da878fa47","resolution":{"observed_at":"2026-08-07T13:44:35.245526Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1811.00937","last_updated":"2019-03-15T18:02:58Z","snapshot_observed_at":"2026-08-10T07:37:55.457401Z","submitted_at":"2018-11-02T15:34:29Z","title":"CommonsenseQA: A Question Answering Challenge Targeting Commonsense Knowledge","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1811.00937","snapshot_observed_at":"2026-08-07T13:44:35.393197Z","title":"Commonsenseqa: A ques- tion answering challenge targeting commonsense knowledge.arXiv preprint arXiv:1811.00937, 2018","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:35.393197Z"},"links":{"cited_paper":"/paper/1811.00937","citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:0333bf30471557320aa72d50442da51e3fe23965b5f6b904ae87cbd8cfd9240c","observation_id":"b91491c9-ebee-4694-bcdc-98a6d2ef8eea","resolution":{"observed_at":"2026-08-07T13:44:35.393197Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:44:38.203090Z","title":"Unlocking the potential of model merging for low-resource languages","venue":null,"work_id":"d7d5add1-21b9-4ac0-b36c-0c7ebac7a0f3","year":2024},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:35.501545Z"},"links":{"citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:c5ae8aaa32aed594e8f54a764b89b054be8005f52d57f63a6d168325d1c81fb9","observation_id":"edba27e7-7af2-41c2-8b25-c472e8ae6012","resolution":{"observed_at":"2026-08-07T13:44:38.282317Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00118","last_updated":"2024-10-02T15:22:49Z","snapshot_observed_at":"2026-08-02T16:20:09.773989Z","submitted_at":"2024-07-31T19:13:07Z","title":"Gemma 2: Improving Open Language Models at a Practical Size","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00118","snapshot_observed_at":"2026-08-07T13:44:35.591163Z","title":"Gemma 2: Improving open language models at a practical size.arXiv preprint arXiv:2408.00118, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:35.591163Z"},"links":{"cited_paper":"/paper/2408.00118","citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:7d7e098d434d4b28247bc13b90882ff4add89714e06334685703e25617252a14","observation_id":"2ee1fe58-fe4d-47ae-994f-d2886fcc51d7","resolution":{"observed_at":"2026-08-07T13:44:35.591163Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:44:38.056049Z","title":"Estimation in high dimensions: a geometric perspective","venue":null,"work_id":"66abdc70-c1cb-41ed-b5c8-a4f6728480a3","year":2015},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:35.688556Z"},"links":{"citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:ec7756436972b0187ee63a3836178b43c068fde15cb230a6e842409be6ce887f","observation_id":"a78f64a5-2263-436b-97be-29d03aaa3b5a","resolution":{"observed_at":"2026-08-07T13:44:38.115696Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:44:37.869154Z","title":"Principal component analysis.Chemometrics and intelligent laboratory systems, 2(1-3):37–52, 1987","venue":null,"work_id":"06090968-bc47-40ec-a81e-c410de24de5b","year":1987},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:35.866143Z"},"links":{"citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:c8bbbd2597d85ef49fb28e35c46918bf50252087d2969a16f108353a6df2b569","observation_id":"6038e34c-c053-442d-8403-119bcdca394e","resolution":{"observed_at":"2026-08-07T13:44:37.954937Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:44:35.949478Z","title":"Ties-merging: Resolving interference when merging models.Advances in Neural Information Processing Systems, 36:7093–7115, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:35.949478Z"},"links":{"citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:e2ca0b0c772e902bfe3f71c63d1808aae264bdce31a570a35489dd93818e097f","observation_id":"713dba1e-8030-4202-a7d1-c82ead5ede4c","resolution":{"observed_at":"2026-08-07T13:44:35.949478Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03617","last_updated":"2024-10-04T17:17:19Z","snapshot_observed_at":"2026-08-12T22:30:33.271257Z","submitted_at":"2024-10-04T17:17:19Z","title":"What Matters for Model Merging at Scale?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.03617","snapshot_observed_at":"2026-08-07T13:44:36.048280Z","title":"What matters for model merging at scale?arXiv preprint arXiv:2410.03617, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:36.048280Z"},"links":{"cited_paper":"/paper/2410.03617","citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:c744068b1c3c9e9d0dc1a52ffb70394533947d037696817d7be37ae4da693861","observation_id":"d75f9e00-ade1-4bfa-87a9-8e15c2c7303c","resolution":{"observed_at":"2026-08-07T13:44:36.048280Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02575","last_updated":"2024-05-28T06:42:31Z","snapshot_observed_at":"2026-08-13T05:57:54.020801Z","submitted_at":"2023-10-04T04:26:33Z","title":"AdaMerging: Adaptive Model Merging for Multi-Task Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.02575","snapshot_observed_at":"2026-08-07T13:44:36.149510Z","title":"Adamerging: Adaptive model merging for multi-task learning.arXiv preprint arXiv:2310.02575, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:36.149510Z"},"links":{"cited_paper":"/paper/2310.02575","citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:f089ddde26978aaffeb0b486c4972f12d06b19e0259412d3b4e884f5d163de76","observation_id":"b7b36f27-715f-4f4d-8a43-5462faa7c56e","resolution":{"observed_at":"2026-08-07T13:44:36.149510Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:44:36.280683Z","title":"Language models are super mario: Absorbing abilities from homologous models as a free lunch","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:36.280683Z"},"links":{"citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:750fe72cde6f0bfdd2ae3e80d2bf4b5a8b81735927f8914e6c3b11389eab942e","observation_id":"e76b9d63-4e76-4c68-8ac7-2615fb08f8dd","resolution":{"observed_at":"2026-08-07T13:44:36.280683Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:44:37.682900Z","title":"Emotion detection on tv show transcripts with sequence- based convolutional neural networks","venue":null,"work_id":"42fc08de-d456-45f6-860f-2686a3d202f8","year":2018},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:36.447501Z"},"links":{"citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:39384cd95021613b38370eaa4e6d783be9a43c5faaa2701566a3e5ebc9361731","observation_id":"40992453-269c-43b9-915f-861568f81fa5","resolution":{"observed_at":"2026-08-07T13:44:37.766416Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:44:36.543531Z","title":"Composing parameter-efficient modules with arithmetic operation.Advances in Neural Information Processing Systems, 36:12589–12610, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:36.543531Z"},"links":{"citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:e13f84107425d9471bb6e23388c87f827df019b476981e841ba46fb085b9287d","observation_id":"3ecc682e-1cf1-4fd6-bd56-cee4ab56e020","resolution":{"observed_at":"2026-08-07T13:44:36.543531Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01155","last_updated":"2025-03-03T04:03:31Z","snapshot_observed_at":"2026-08-07T17:34:57.455440Z","submitted_at":"2025-03-03T04:03:31Z","title":"Nature-Inspired Population-Based Evolution of Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.01155","snapshot_observed_at":"2026-08-07T13:44:36.640223Z","title":"Nature-inspired population-based evolution of large language models.arXiv preprint arXiv:2503.01155, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:36.640223Z"},"links":{"cited_paper":"/paper/2503.01155","citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:43c4def2819b8567fe5b6e1bc2e94fe96ce11fc106e77222469f08a5c2fb125d","observation_id":"e151f6e4-f5a9-48ce-8208-125b4414a467","resolution":{"observed_at":"2026-08-07T13:44:36.640223Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.13372","last_updated":"2024-06-27T22:44:48Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-03-20T08:08:54Z","title":"LlamaFactory: Unified Efficient Fine-Tuning of 100+ Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.13372","snapshot_observed_at":"2026-08-07T13:44:36.776858Z","title":"Llamafactory: Unified efficient fine-tuning of 100+ language models.arXiv preprint arXiv:2403.13372, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:36.776858Z"},"links":{"cited_paper":"/paper/2403.13372","citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:07ec3ef3cc348db4db4b68a5e2c4ca2ef41d4d09ecd1cf29861cae49e900b52f","observation_id":"ae8bc4e2-e34d-47ee-aacc-744cc08cdf91","resolution":{"observed_at":"2026-08-07T13:44:36.776858Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:44:37.513153Z","title":"particle","venue":null,"work_id":"2a7e689a-d839-476a-8e4b-7f19553e2aa6","year":2021},"citing_paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T13:44:36.927598Z"},"links":{"citing_paper":"/paper/2505.21226"},"observation_digest":"sha256:8817772f61efec1f4b3c8ae3bc5541575807f23aa4e8b86268174536057be290","observation_id":"b5c8c531-4c32-486b-8dce-fc8bdfb86d45","resolution":{"observed_at":"2026-08-07T13:44:37.554033Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2505.21226","last_updated":"2025-06-03T14:43:50Z","latest_version":2,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-09T03:32:40.483446Z","submitted_at":"2025-05-27T14:10:46Z","title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging"},"reference_resolution":{"displayed":38,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":30,"verified_exact":1,"verified_fuzzy":7},"total_outbound_references":38},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"thesis":"As of 13 August 2026, this Paper Citation Record lists 38 of 38 outbound references and 5 inbound Pith citation observations for arXiv:2505.21226."}